From 903b9bf1873fa52daf9d0000c83ac7ffeb28fc4a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:10:26 -0700 Subject: [PATCH 001/276] =?UTF-8?q?fix(delegate):=20nested=20orchestrators?= =?UTF-8?q?=20get=20their=20workers'=20results=20back=20=E2=80=94=20no=204?= =?UTF-8?q?20=20s=20deadline=20on=20delegate=5Ftask,=20summary=20budget=20?= =?UTF-8?q?uses=20the=20current=20prompt=20not=20the=20session=20sum?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two defects in the same path, both measured on the 1,393-agent refactor run. 1. A nested orchestrator (depth > 0) runs delegate_task synchronously by design: it needs its workers' results inside its own turn. But the sequential tool runner put every tool call under the generic 420 s deadline, and delegate_task was not exempt, so every batch longer than seven minutes returned "timed out after 420.0s" while the children kept running as orphans. 332 such timeouts in 234 orchestrator sessions; only 89 nested delegate_task calls in the whole run ever returned a real result. The orchestrators then spent 388 h of wall time polling: 1,526 reads of the live transcript files, 551 list actions, 242 h of explicit sleep, about $4k of API turns. delegate_task is now exempt from the sequential deadline (the batch owns its liveness: per-child heartbeats, the stale monitor, delegation.child_timeout_seconds). Live A/B, depth-1 orchestrator dispatching a 75 s leaf with the deadline set to 40 s (glm-5.3-flash via Nous): main -> "Error executing tool 'delegate_task': timed out after 40.0s", leaf result lost; branch -> orchestrator blocked 161 s and returned the leaf's LEAF_DONE_MARKER. 2. _parent_summary_char_budget computed the parent's context headroom from session_prompt_tokens, which is the running SUM of prompt tokens over every API call in the session. After a few hundred calls it exceeds any window, headroom goes negative, and every child summary collapses to the 2,000-char floor with the full text spilled to disk. All 1,393 child summaries in the run were truncated this way; the orchestrator planned from stubs. The budget now reads the last call's prompt_tokens from _last_turn_usage. Tests: delegate_task is in the exempt set and the set is narrow; budget for a long-lived parent equals the budget for a fresh parent with the same current prompt, and exceeds the floor. --- agent/tool_executor.py | 11 ++++++++++- .../test_sequential_deadline_delegate_exempt.py | 13 +++++++++++++ tests/tools/test_delegate_summary_budget.py | 16 ++++++++++++++-- tools/delegate_tool_results.py | 15 +++++++++++---- 4 files changed, 48 insertions(+), 7 deletions(-) create mode 100644 tests/agent/test_sequential_deadline_delegate_exempt.py diff --git a/agent/tool_executor.py b/agent/tool_executor.py index a7fbd13384..799199c10f 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -769,6 +769,15 @@ def _resolve_sequential_tool_timeout() -> float | None: return resolve_timeout("tools.sequential_call", default=_resolve_concurrent_tool_timeout()) +# Tools whose call blocks on a long-running operation that supervises its own liveness: no generic +# sequential deadline. ``delegate_task`` in a nested orchestrator blocks for the whole batch by design +# (children carry heartbeats, the stale monitor, and ``delegation.child_timeout_seconds``); under the +# 420 s deadline every real batch "timed out" while its children ran on as orphans, and the orchestrator +# spent the following hours polling transcripts (measured: 332 timeouts, ~$4k of orchestrator turns in +# one run). +_SEQUENTIAL_DEADLINE_EXEMPT_TOOLS = frozenset({"delegate_task"}) + + def _abandoned_sequential_result(agent, ref: _ToolCallRef, message: str, result_cls, **outcome) -> _ManagedToolResult: """Emit the terminal post_tool_call for a worker the sequential runner gave up on (timeout / interrupt) and wrap ``message`` in its marker ``result_cls``.""" @@ -815,7 +824,7 @@ def _run_sequential_tool_execution_middleware( """Run one sequential call on a worker thread under the concurrent executor's deadline. Interactive tools (``clarify``) own their wait via ``agent.clarify_timeout``; the generic deadline would report ``tool_timeout`` while the prompt is still live.""" - timeout_s = _resolve_sequential_tool_timeout() + timeout_s = None if function_name in _SEQUENTIAL_DEADLINE_EXEMPT_TOOLS else _resolve_sequential_tool_timeout() ref = _ToolCallRef(function_name, function_args, effective_task_id, tool_call_id, middleware_trace) kwargs = dict(ref.middleware_kwargs(), execute=execute, scope_block=scope_block, display_index=display_index) if function_name in _NEVER_PARALLEL_TOOLS: diff --git a/tests/agent/test_sequential_deadline_delegate_exempt.py b/tests/agent/test_sequential_deadline_delegate_exempt.py new file mode 100644 index 0000000000..3b881c3904 --- /dev/null +++ b/tests/agent/test_sequential_deadline_delegate_exempt.py @@ -0,0 +1,13 @@ +"""``delegate_task`` runs a nested orchestrator's whole batch inside one tool call by design, so it must not +sit under the generic sequential-call deadline: with it, every batch longer than the deadline "timed out" +while its children kept running as orphans and the orchestrator polled transcripts for hours.""" +from agent import tool_executor as te + + +def test_delegate_task_is_exempt_from_the_sequential_deadline(): + assert "delegate_task" in te._SEQUENTIAL_DEADLINE_EXEMPT_TOOLS + + +def test_exemption_is_narrow(): + assert "terminal" not in te._SEQUENTIAL_DEADLINE_EXEMPT_TOOLS + assert "execute_code" not in te._SEQUENTIAL_DEADLINE_EXEMPT_TOOLS diff --git a/tests/tools/test_delegate_summary_budget.py b/tests/tools/test_delegate_summary_budget.py index a9d826dee6..f5b81fed85 100644 --- a/tests/tools/test_delegate_summary_budget.py +++ b/tests/tools/test_delegate_summary_budget.py @@ -13,6 +13,7 @@ import tempfile import pytest import tools.delegate_tool as dt +from tools.delegate_tool_results import _MIN_SUMMARY_CHARS, _parent_summary_char_budget class _FakeCompressor: @@ -22,9 +23,11 @@ class _FakeCompressor: class _FakeParent: - def __init__(self, context_length, used_tokens, max_tokens): + def __init__(self, context_length, used_tokens, max_tokens, session_total=None): self.context_compressor = _FakeCompressor(context_length, max_tokens) - self.session_prompt_tokens = used_tokens + # Current prompt size (last call) drives the budget; the cumulative session counter must not. + self._last_turn_usage = {"prompt_tokens": used_tokens} + self.session_prompt_tokens = session_total if session_total is not None else used_tokens def test_small_summaries_pass_through_untouched(): @@ -76,3 +79,12 @@ def test_empty_results_is_noop(): [{"task_index": 0, "status": "failed", "summary": None}], _FakeParent(131_000, 1_000, 8_000), ) + + +def test_budget_uses_current_prompt_size_not_the_session_sum(): + """A long-lived parent has a session sum far past any window while its current prompt is small; the + budget must follow the current prompt, otherwise every summary collapses to the floor.""" + long_lived = _FakeParent(context_length=200_000, used_tokens=30_000, max_tokens=8_000, session_total=25_000_000) + fresh = _FakeParent(context_length=200_000, used_tokens=30_000, max_tokens=8_000) + assert _parent_summary_char_budget(long_lived, 1) == _parent_summary_char_budget(fresh, 1) + assert _parent_summary_char_budget(long_lived, 1) > _MIN_SUMMARY_CHARS diff --git a/tools/delegate_tool_results.py b/tools/delegate_tool_results.py index c3b08a58be..f45e7c13ee 100644 --- a/tools/delegate_tool_results.py +++ b/tools/delegate_tool_results.py @@ -219,15 +219,22 @@ def _trim_summary_with_footer(summary: str, cap: int, task_index: int) -> tuple[ return head + "\n\n[... middle omitted — see footer ...]\n\n" + tail + "\n".join(footer_lines), spill_path def _parent_summary_char_budget(parent_agent, n_summaries: int) -> Optional[int]: - """Per-summary char budget from the parent's *remaining* context headroom (context length − prompt tokens − the - compressor's output reserve), a fraction of it split across the batch at ~4 chars/token. None when the parent's - context state is unknown — caller then uses the static ceiling only.""" + """Per-summary char budget from the parent's *remaining* context headroom (context length − the parent's + current prompt size − the compressor's output reserve), a fraction of it split across the batch at ~4 + chars/token. None when the parent's context state is unknown — caller then uses the static ceiling only. + + "Current prompt size" is the last API call's ``prompt_tokens`` (``_last_turn_usage``), never + ``session_prompt_tokens``: that field is the running SUM over every call in the session, so after a + few hundred calls it exceeds any context window and every summary collapsed to the 2,000-char floor + (measured: all 1,393 child summaries in one run, each spilled to disk, the orchestrator working from + the stub).""" try: compressor = getattr(parent_agent, "context_compressor", None) context_length = getattr(compressor, "context_length", None) if not isinstance(context_length, int) or context_length <= 0: return None - used_tokens = getattr(parent_agent, "session_prompt_tokens", 0) + last_usage = getattr(parent_agent, "_last_turn_usage", None) or {} + used_tokens = last_usage.get("prompt_tokens") if isinstance(last_usage, dict) else None if not isinstance(used_tokens, (int, float)) or used_tokens < 0: used_tokens = 0 headroom_tokens = context_length - int(used_tokens) - int(getattr(compressor, "max_tokens", 0) or 0) From d41f62107148b630986521bc1286f575598a3aa6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:19:03 -0700 Subject: [PATCH 002/276] fix(approval): a grep inside "$(...)" no longer trips the hardline "malformed payload" block The quoted-grep scanner tokenizes from the grep's position to decide which quoted operand is inert data. _shell_tokens_with_spans lexed from there to the END of the top-level segment, so for a grep nested in a substitution such as sed -n "$(grep -n X f | cut -d: -f1),+3p" f it read past the closing ")" into the enclosing command, met the outer closing double quote as an opener, ended in quote state, returned None ("malformed"), and detect_hardline_command reported the whole command as being on the unconditional blocklist. Nothing about the command was dangerous. In the 1,393-agent refactor run this fired 546 times across the workers (every one a false positive; 318 of 323 retries succeeded by rewording), and three times on the maintainer's own session in one day. The lexer now stops at the end of the simple command it was asked to lex: an unquoted list separator or pipe or newline, or the ")" / backtick that closes the substitution it sits inside (tracking $(...) depth opened after start so a nested substitution's closer is not mistaken for the outer one). Genuinely unbalanced quoting still returns None and still fails closed; every hardline pattern is unchanged. Tests: six substitution shapes lex clean and are not hardline; the lexer yields exactly the grep's own words; unterminated quoting still blocks; rm -rf / , $(rm -rf /) and shutdown still block. --- .../test_approval_grep_in_substitution.py | 41 +++++++++++++++++++ tools/approval_detection.py | 41 ++++++++++++++----- 2 files changed, 72 insertions(+), 10 deletions(-) create mode 100644 tests/tools/test_approval_grep_in_substitution.py diff --git a/tests/tools/test_approval_grep_in_substitution.py b/tests/tools/test_approval_grep_in_substitution.py new file mode 100644 index 0000000000..abfff1359b --- /dev/null +++ b/tests/tools/test_approval_grep_in_substitution.py @@ -0,0 +1,41 @@ +"""A grep nested inside a command substitution must lex as its own simple command. + +The quoted-grep scanner tokenizes from the grep's position to decide which quoted operand is inert data. +It used to read to the end of the segment, so for ``"$(grep … | cut …)"`` it swallowed the enclosing +command's closing quote, called the quoting unbalanced, and the whole command was reported as a hardline +block. Every such report was a false positive (546 in one run); the lexer now stops at the end of the +simple command: an unquoted ``;`` ``|`` ``&`` newline, or the ``)`` / backtick closing its substitution. +""" +import pytest + +from tools.approval_detection import _quoted_grep_pattern_spans, _shell_tokens_with_spans, detect_hardline_command + + +@pytest.mark.parametrize("command", [ + 'sed -n "$(grep -n X f | cut -d: -f1),+3p" f', + 'sed -n "$(grep -n \'^### 1.\' d.md | cut -d: -f1),$(grep -n \'^### 3.\' d.md | cut -d: -f1)p" d.md', + 'echo "$(grep -c x f)"', + 'x="$(grep -n foo bar | head -1)"; echo $x', + 'for f in a b; do echo "$f sig=$(grep -cE \'^x\' $f)"; done', + 'echo `grep -c x f`', +]) +def test_grep_inside_a_substitution_is_not_malformed(command): + _spans, malformed = _quoted_grep_pattern_spans(command) + assert malformed is False + assert detect_hardline_command(command) == (False, None) + + +def test_lexer_stops_at_the_end_of_the_simple_command(): + seg = 'sed -n "$(grep -n X f | cut -d: -f1),+3p" f' + toks = _shell_tokens_with_spans(seg, seg.index("grep")) + assert [t[0] for t in toks] == ["grep", "-n", "X", "f"] + + +def test_genuinely_unbalanced_quoting_still_fails_closed(): + assert _shell_tokens_with_spans("grep 'unterminated", 0) is None + assert detect_hardline_command("grep 'unterminated")[0] is True + + +def test_hardline_patterns_unchanged(): + for command in ("rm -rf /", 'echo "$(rm -rf /)"', "sudo shutdown -h now"): + assert detect_hardline_command(command)[0] is True, command diff --git a/tools/approval_detection.py b/tools/approval_detection.py index 1ef703670a..23b7613abd 100644 --- a/tools/approval_detection.py +++ b/tools/approval_detection.py @@ -581,8 +581,16 @@ def _command_parser_limit_exceeded(command: str) -> bool: def _shell_tokens_with_spans(segment: str, start: int): """Return shell words as ``(value, start, end, quoted)`` or ``None`` on malformed quoting. Deliberately small lexer that never expands shell syntax; it exists to keep source spans (which - ``shlex`` does not expose) for deciding which quoted grep operand is data, not another command.""" + ``shlex`` does not expose) for deciding which quoted grep operand is data, not another command. + + Lexing stops at the end of the simple command that begins at *start*: an unquoted ``;``, ``|``, + ``&`` or newline, or the ``)`` / backtick that closes the substitution the command sits inside. + Without that, a grep nested as ``"$(grep … | cut …)"`` was lexed together with the enclosing + command's closing quote, read as unbalanced quoting, and reported as a hardline block (546 + blocked turns in one run, every one a false positive; ``sed -n "$(grep -n X f | cut -d: -f1),+3p" f`` + is the canonical shape).""" tokens, value, token_start, quote = [], [], None, None + depth = 0 # $(...) nesting opened AFTER start; a closer at depth 0 ends the enclosing substitution def flush(end: int) -> None: raw = segment[token_start:end] @@ -591,26 +599,39 @@ def _shell_tokens_with_spans(segment: str, start: int): inert = (raw.startswith("'") and raw.endswith("'")) or ("='" in raw and raw.endswith("'")) tokens.append(("".join(value), token_start, end, inert)) + end_at = len(segment) for kind, i, _, _ in _scan_shell(segment, start): - if kind == "char" and not quote and segment[i].isspace(): - if token_start is not None: - flush(i) - value, token_start = [], None - continue + ch = segment[i] + if kind == "char" and not quote: + if ch.isspace() and ch != "\n": + if token_start is not None: + flush(i) + value, token_start = [], None + continue + if segment.startswith("$(", i): + depth += 1 + elif ch == ")" or ch == "`": + if depth == 0: + end_at = i + break + depth -= 1 + elif ch in ";|&\n": + end_at = i + break if token_start is None: token_start = i if kind == "quote": - quote = None if quote else segment[i] + quote = None if quote else ch elif kind == "esc": value.append(segment[i + 1]) - elif segment[i] == "\\" and not quote: + elif ch == "\\" and not quote: return None # dangling backslash else: - value.append(segment[i]) + value.append(ch) if quote: return None if token_start is not None: - flush(len(segment)) + flush(end_at) return tokens From f8b87f5637653ca64391e70248d497ab5ec7efc6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:30:19 -0700 Subject: [PATCH 003/276] fix(goal): the judge only sees the goal's own background processes, and a pid/session wait barrier expires after 30 min In a fan-out run the /goal loop parked for 3 h 22 min at the end on a grandchild's poller (waiting_on_session=proc_21a6fe2369a1, 00:47 -> 04:09) while the root itself had nothing running. Every one of the root's 7 logged judge verdicts was WAIT on a child-owned process. Two causes. gather_background_processes() was called with no task_id from both goal-loop callers (CLI cli_loops_mixin, gateway run_goals), and the registry's task_id is the CONTAINER key, which collapses to one value for every agent in the process, so the judge's process list was every one of ~1,300 subagents' pollers. And a pid/session barrier had no ceiling: once the judge said WAIT on a session that never exits, nothing resumed judging; waiting_since was recorded and never read. Now list_sessions() reports owner_task_id (the RAW spawning id the registry already keeps for ownership checks), gather_background_processes takes owner_task_id and both callers pass their own session id (CLI turns register processes under self.session_id; gateway turns under turn_ctx.session_id), and is_waiting() ages out a pid/session barrier after _MAX_BARRIER_WAIT_S (30 min). Timed barriers keep their own deadline. Tests: only the owning session's running processes are returned when owner_task_id is given (unfiltered behaviour unchanged); a live barrier older than the ceiling clears and judging resumes. --- gateway/run_goals.py | 4 +++- hermes_cli/cli_loops_mixin.py | 3 ++- hermes_cli/goals.py | 27 +++++++++++++++++++---- tests/hermes_cli/test_goals.py | 40 ++++++++++++++++++++++++++++++++++ tools/process_registry.py | 1 + 5 files changed, 69 insertions(+), 6 deletions(-) diff --git a/gateway/run_goals.py b/gateway/run_goals.py index ef2bd26d41..011fc12c11 100644 --- a/gateway/run_goals.py +++ b/gateway/run_goals.py @@ -244,7 +244,9 @@ class GatewayGoalsMixin: _bg_procs = None with suppress(Exception): from hermes_cli.goals import gather_background_processes as _gather_bg - _bg_procs = _gather_bg() + # Only THIS session's processes (gateway turns register under turn_ctx.session_id): + # subagents' pollers must not park the parent's goal. + _bg_procs = _gather_bg(owner_task_id=getattr(session_entry, "session_id", None) or None) # judge_goal() is a synchronous aux-LLM HTTP call (10-40 s; would block Discord heartbeats). # _run_in_executor_with_context carries the profile secret scope / aux runtime contextvars diff --git a/hermes_cli/cli_loops_mixin.py b/hermes_cli/cli_loops_mixin.py index fa00827bff..4622a60561 100644 --- a/hermes_cli/cli_loops_mixin.py +++ b/hermes_cli/cli_loops_mixin.py @@ -533,7 +533,8 @@ class CLILoopsMixin: return try: from hermes_cli.goals import gather_background_processes as _gather_bg - _bg_procs = _gather_bg() + # Only THIS session's processes: subagents' pollers must not park the parent's goal. + _bg_procs = _gather_bg(owner_task_id=getattr(self, "session_id", None) or None) except Exception: _bg_procs = None decision = mgr.evaluate_after_turn( diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py index 8d0b24f679..d730230712 100644 --- a/hermes_cli/goals.py +++ b/hermes_cli/goals.py @@ -49,6 +49,9 @@ DEFAULT_MAX_CONSECUTIVE_TRANSPORT_FAILURES = 5 # on concrete evidence instead of a vibe check. DEFAULT_GATE_TIMEOUT_SECONDS = 300 DEFAULT_GATE_MAX_RETRIES = 3 +# Longest a pid/session wait barrier may hold the loop before judging resumes. Timed barriers +# (``waiting_until``) carry their own deadline and are exempt. +_MAX_BARRIER_WAIT_S = 30 * 60 # Bounded tail of a failed gate's combined stdout/stderr fed back to the agent. _GATE_OUTPUT_TAIL_CHARS = 3000 @@ -909,9 +912,15 @@ def judge_goal( return verdict, reason, parse_failed, wait_directive, False -def gather_background_processes(task_id: Optional[str] = None) -> List[Dict[str, Any]]: +def gather_background_processes(task_id: Optional[str] = None, *, owner_task_id: Optional[str] = None) -> List[Dict[str, Any]]: """Fail-safe snapshot of RUNNING ``process_registry`` sessions for the judge; ``[]`` on any error - so the loop degrades to its pre-wait-barrier behavior.""" + so the loop degrades to its pre-wait-barrier behavior. + + ``owner_task_id`` restricts the snapshot to processes the goal's OWN session spawned. The registry's + ``task_id`` is the container key, which collapses to one value for every agent in the process, so + without this filter a fan-out parent's judge saw every subagent's pollers and parked the goal on a + grandchild's ``proc_*`` session (one run: 7 of 7 root verdicts were WAIT on child-owned processes; + parked 3 h 22 min at the end while nothing of its own was running).""" try: from tools.process_registry import process_registry @@ -919,7 +928,10 @@ def gather_background_processes(task_id: Optional[str] = None) -> List[Dict[str, except Exception as exc: logger.debug("gather_background_processes failed: %s", exc) return [] - return [s for s in sessions if isinstance(s, dict) and s.get("status") != "exited"] + running = [s for s in sessions if isinstance(s, dict) and s.get("status") != "exited"] + if owner_task_id: + running = [s for s in running if str(s.get("owner_task_id") or s.get("task_id") or "") == str(owner_task_id)] + return running def draft_contract(objective: str, *, timeout: Optional[float] = None) -> Optional[GoalContract]: @@ -1282,7 +1294,10 @@ class GoalManager: def is_waiting(self) -> bool: """True iff a barrier is set AND not yet satisfied. A satisfied barrier is cleared here - (lazy auto-clear) so the next evaluation resumes normal judging.""" + (lazy auto-clear) so the next evaluation resumes normal judging. A pid/session barrier + also expires after ``_MAX_BARRIER_WAIT_S``: a watcher or poller that never exits would + otherwise park the goal indefinitely (one run sat 3 h 22 min on a poller that outlived + the work it was polling).""" s = self._state if s is None: return False @@ -1294,6 +1309,10 @@ class GoalManager: still = time.time() < s.waiting_until else: return False + if still and s.waiting_since and s.waiting_until == 0.0 and time.time() - s.waiting_since > _MAX_BARRIER_WAIT_S: + logger.info("goal %s: wait barrier on %s exceeded %ds; resuming judging", + self.session_id, s.waiting_on_session or s.waiting_on_pid, _MAX_BARRIER_WAIT_S) + still = False if not still: self.stop_waiting() return still diff --git a/tests/hermes_cli/test_goals.py b/tests/hermes_cli/test_goals.py index 413a9330ed..f3b2ea7439 100644 --- a/tests/hermes_cli/test_goals.py +++ b/tests/hermes_cli/test_goals.py @@ -436,6 +436,46 @@ class TestWaitBarrier: proc.terminate() proc.wait(timeout=10) + def test_barrier_on_a_process_that_never_exits_expires(self, hermes_home): + """A poller that outlives the work parked one run for 3h22m; a live barrier ages out.""" + from hermes_cli import goals + from hermes_cli.goals import GoalManager + + proc = self._spawn_sleeper() + try: + mgr = GoalManager(session_id="wb-expire") + mgr.set("g") + mgr.wait_on(proc.pid, reason="poller") + assert mgr.is_waiting() is True + mgr.state.waiting_since = time.time() - goals._MAX_BARRIER_WAIT_S - 1 + mgr._save() + assert mgr.is_waiting() is False + assert mgr.state.waiting_on_pid is None + finally: + proc.terminate() + proc.wait(timeout=10) + + +class TestGatherBackgroundProcessesOwnership: + def test_only_the_owning_sessions_processes_are_seen(self, monkeypatch): + """The registry task_id collapses to one container key for every agent in the process, so a + fan-out parent's judge must filter by owner; otherwise a grandchild's poller parks the goal.""" + from hermes_cli import goals + + class _Reg: + def list_sessions(self, task_id=None, session_key=None): + return [ + {"session_id": "proc_mine", "status": "running", "owner_task_id": "root-sid", "task_id": "default"}, + {"session_id": "proc_child", "status": "running", "owner_task_id": "sa-1-abc", "task_id": "default"}, + {"session_id": "proc_done", "status": "exited", "owner_task_id": "root-sid", "task_id": "default"}, + ] + + import tools.process_registry as pr + monkeypatch.setattr(pr, "process_registry", _Reg()) + assert [p["session_id"] for p in goals.gather_background_processes()] == ["proc_mine", "proc_child"] + assert [p["session_id"] for p in goals.gather_background_processes(owner_task_id="root-sid")] == ["proc_mine"] + + # ────────────────────────────────────────────────────────────────────── # Judge-driven auto-wait — the judge parks the loop on its own diff --git a/tools/process_registry.py b/tools/process_registry.py index 75ae003b84..bc4ed7a419 100644 --- a/tools/process_registry.py +++ b/tools/process_registry.py @@ -1776,6 +1776,7 @@ class ProcessRegistry: "command": s.command[:200], "cwd": s.cwd, "pid": s.pid, + "owner_task_id": s.owner_task_id or s.task_id, "started_at": time.strftime("%Y-%m-%dT%H:%M:%S", time.localtime(s.started_at)), "uptime_seconds": int(time.time() - s.started_at), "status": "exited" if s.exited else "running", From f5f54ab7846d780ba2f6ca69f9f58cd55f4a3cbc Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:30:41 -0700 Subject: [PATCH 004/276] docs(goals): the judge sees only the session's own processes; pid/session waits cap at 30 min --- website/docs/user-guide/features/goals.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/website/docs/user-guide/features/goals.md b/website/docs/user-guide/features/goals.md index d187dd9bef..40949e6493 100644 --- a/website/docs/user-guide/features/goals.md +++ b/website/docs/user-guide/features/goals.md @@ -152,7 +152,7 @@ Gates and contracts compose: use a contract to shape *what the agent aims for*, Some goals are gated on something that takes minutes and runs on its own — CI on a pushed PR, a long build, a test matrix, a deploy, a rate-limit cooldown. Without help, the goal loop would re-poke the agent every turn into "is it done yet?" busy-work while it waits. -**This is handled automatically.** Every turn, the judge is shown the agent's live background processes (the `terminal(background=true)` registry — pid, session id, command, uptime, recent output, and any `watch_patterns` / `notify_on_complete` trigger) alongside the goal and the agent's response. When the agent's progress is genuinely gated on one of them, the judge returns a **`wait`** verdict instead of `continue`, and the loop **parks**: the next turns are skipped (no judge call, no continuation, no turn consumed) until the wait is satisfied — then it resumes normally with the result in hand. The judge can also park on a **time** basis (`wait_for_seconds`) for backoff/cooldown waits. `/goal status` shows `⏳ Goal (parked …)` while parked. +**This is handled automatically.** Every turn, the judge is shown the agent's own live background processes (the `terminal(background=true)` registry entries this session spawned — pid, session id, command, uptime, recent output, and any `watch_patterns` / `notify_on_complete` trigger; processes started by delegated subagents are not shown, so a fan-out parent is never parked on a worker's poller) alongside the goal and the agent's response. When the agent's progress is genuinely gated on one of them, the judge returns a **`wait`** verdict instead of `continue`, and the loop **parks**: the next turns are skipped (no judge call, no continuation, no turn consumed) until the wait is satisfied — then it resumes normally with the result in hand. A pid/session wait is capped at 30 minutes; a process that never exits (a watcher, a forgotten poller) cannot park the goal indefinitely. The judge can also park on a **time** basis (`wait_for_seconds`) for backoff/cooldown waits. `/goal status` shows `⏳ Goal (parked …)` while parked. The judge picks the right kind of wait from the process's own signal: From 40da0fd52f018199571841ccb2618d6b3ac8ed22 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 01:07:09 -0700 Subject: [PATCH 005/276] feat(delegation): subagents compress at an absolute context cap (delegation.compression_threshold_tokens, default 200K) A delegate_task child inherits the compression threshold as a RATIO of the model window. On a 1M-window model at the run's configured 0.85 that is an 850K-token trigger: in the 1,393-agent refactor run 1,373 of 1,375 children never compressed once, 62% of all API calls carried >150K of context, and the calls above 200K carried ~$10.9k of the $19.3k bill (58% of it cache WRITES, i.e. re-sending a 300-800K prefix on every call). A sawtooth replay of the logged calls with a 200K cap / 65K floor cuts context spend by ~49% (~$7k). Children are brief-driven and disposable; they re-read their brief and the files they touch, so a large window buys them little. New delegation.compression_threshold_tokens (default 200000) is applied to the child's ContextCompressor right after construction as the lower of it and any global compression.threshold_tokens; the parent's own trigger is untouched. 0 disables the subagent-specific cap. The compressor applies threshold_tokens_cap on first window resolution, so this is byte-equivalent to the user having set compression.threshold_tokens for the child. Live through the real spawn path (_build_child_agent, real imports, temp HERMES_HOME, 1M-window model): main child trigger 500,000 / branch 200,000; parent 500,000 on both. Tests (3): default caps a 1M child at 200K; the cap is the lower of the delegation and global values and never raises a small-window child's trigger; 0 disables and an already-resolved trigger is re-clamped. Docs: delegation.md, configuration.md. --- hermes_cli/config_defaults.py | 7 ++++ .../test_delegate_child_compression_cap.py | 40 +++++++++++++++++++ tools/delegate_tool.py | 22 ++++++++++ website/docs/user-guide/configuration.md | 1 + .../docs/user-guide/features/delegation.md | 2 + 5 files changed, 72 insertions(+) create mode 100644 tests/tools/test_delegate_child_compression_cap.py diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 234ef5dcc1..8ccb38cf30 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1216,6 +1216,13 @@ DEFAULT_CONFIG = { # {"extra_body": {"provider": {"sort": "throughput"}}}. Explicit values win OVER # runtime/parent overrides (extra_body deep-merged 1 level). "request_overrides": {}, + # compression_threshold_tokens: absolute context cap for subagents, applied as the lower of + # this and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window + # model otherwise compresses at threshold x window (850K at 0.85) and re-sends a 300-800K + # prefix on every call: in one 1,393-agent run 1,373 children never compressed and calls + # above 200K context carried ~55% of the bill. Independent of the parent's own threshold. + # 0 = no subagent-specific cap. + "compression_threshold_tokens": 200000, # When delegate_task narrows child toolsets, keep the parent's enabled MCP toolsets (so # toolsets=["web"] doesn't strip MCP). false = strict intersection. "inherit_mcp_toolsets": True, diff --git a/tests/tools/test_delegate_child_compression_cap.py b/tests/tools/test_delegate_child_compression_cap.py new file mode 100644 index 0000000000..024c3f0855 --- /dev/null +++ b/tests/tools/test_delegate_child_compression_cap.py @@ -0,0 +1,40 @@ +"""Subagents spawned by delegate_task compress at an absolute context cap, not threshold x window. + +In a 1,393-agent run on a 1M-window model the children's trigger was 0.85 x 1M = 850K; 1,373 of 1,375 +never compressed and calls above 200K context carried ~55% of the bill. +""" +from types import SimpleNamespace + +from agent.context_compressor import ContextCompressor +from tools.delegate_tool import _apply_child_compression_cap + + +def _child(window=1_000_000, threshold=0.85, cap=None): + cc = ContextCompressor(model="anthropic/claude-fable-5.1", threshold_percent=threshold, + config_context_length=window, threshold_tokens_cap=cap) + return SimpleNamespace(context_compressor=cc) + + +def test_default_cap_bounds_a_big_window_child(): + child = _child() + _apply_child_compression_cap(child, {}) + assert child.context_compressor.threshold_tokens == 200_000 + + +def test_cap_is_the_lower_of_delegation_and_global_and_never_raises(): + child = _child(cap=150_000) + _apply_child_compression_cap(child, {"compression_threshold_tokens": 200_000}) + assert child.context_compressor.threshold_tokens == 150_000 + small = _child(window=128_000) + before = small.context_compressor.threshold_tokens + _apply_child_compression_cap(small, {"compression_threshold_tokens": 200_000}) + assert small.context_compressor.threshold_tokens == before # cap above the ratio trigger: no effect + + +def test_zero_disables_and_an_already_resolved_trigger_is_reclamped(): + child = _child() + assert child.context_compressor.threshold_tokens == 850_000 # resolve first + _apply_child_compression_cap(child, {"compression_threshold_tokens": 0}) + assert child.context_compressor.threshold_tokens == 850_000 + _apply_child_compression_cap(child, {"compression_threshold_tokens": 300_000}) + assert child.context_compressor.threshold_tokens == 300_000 diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 70e98fe6a6..c0660647fd 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -102,6 +102,27 @@ def _open_child_session_db(parent_agent) -> Any: return acquire(_parent_db_path) if _parent_db_path is not None else acquire() return None + +def _apply_child_compression_cap(child, delegation_cfg: dict) -> None: + """Cap the child's compaction trigger at ``delegation.compression_threshold_tokens`` (lower of it + and any global ``compression.threshold_tokens``). The compressor applies the cap on first window + resolution, which happens after construction, so setting it here is exactly equivalent to config.""" + from agent.context_compressor import ContextCompressor + + cc = getattr(child, "context_compressor", None) + if not isinstance(cc, ContextCompressor): + return + try: + cap = int((delegation_cfg or {}).get("compression_threshold_tokens", 200_000) or 0) + except (TypeError, ValueError): + cap = 0 + if cap <= 0: + return + existing = cc.threshold_tokens_cap + cc.threshold_tokens_cap = min(cap, existing) if isinstance(existing, int) and existing > 0 else cap + if cc._threshold_tokens is not None: # already resolved: re-clamp now + cc._apply_threshold_tokens_cap() + def _build_child_agent( task_index: int, goal: str, @@ -203,6 +224,7 @@ def _build_child_agent( child._progress_identity_ref = child_session_ref child._delegate_depth, child._delegate_role = child_depth, effective_role # post-degrade role child._subagent_id, child._parent_subagent_id = subagent_id, parent_subagent_id + _apply_child_compression_cap(child, delegation_cfg) # Ownership chain for action=list/steer/stop; weakref so a finished parent # can be collected while a detached child record lingers in the registry. try: diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index a663d1b317..2a18c3c318 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -2636,6 +2636,7 @@ delegation: # base_url: "http://localhost:1234/v1" # Direct OpenAI-compatible endpoint (takes precedence over provider) # api_key: "local-key" # API key for base_url (falls back to OPENAI_API_KEY) # api_mode: "" # Wire protocol for base_url: "chat_completions", "codex_responses", or "anthropic_messages". Empty = auto-detect from URL (e.g. /anthropic suffix → anthropic_messages). Set explicitly for non-standard endpoints the heuristic can't detect. + compression_threshold_tokens: 200000 # Subagent context cap (lower of this and the child's ratio threshold); 0 = none # request_overrides: # Per-child request settings sent on every subagent API call (all resolution branches). # extra_body: # Merged into the request's extra_body — e.g. OpenRouter routing hints: # provider: diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index a4ce8aba3b..dcc51599ac 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -537,6 +537,8 @@ delegation: When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare). +Subagents compress at an absolute cap, `delegation.compression_threshold_tokens` (default `200000`), applied as the lower of it and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window model would otherwise compress at `threshold × window` (850K at 0.85) and re-send a 300–800K prefix on every call; in a 1,393-agent run, 1,373 children never compressed and calls above 200K context carried about 55% of the bill. It is independent of the parent's own threshold; `0` disables the subagent-specific cap. + `delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details. :::tip From 058ad0329e74060ae9dba39b3686b44025055daa Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 01:27:23 -0700 Subject: [PATCH 006/276] fix(nous): adopt a fresh agent key before the one in hand expires, and start the keepalive in every process that routes to Nous The Nous agent key lives 3,599 s. In the 1,393-agent refactor run every in-process agent learned about the hourly expiry from its own 401: 620 authentication_error 401s in the logged window (177 in one hour), each a failed attempt the model never saw, and the credential pool benched the sole credential for all of them at once. At 08:30 the storm took the parent process down. Two gaps. The proactive refresher (hermes_cli/nous_auth_keepalive.py) is started only by the gateway and the web server; the CLI process, and every subagent built inside it, never started it. And even with a fresh key in the store, nothing adopted it before a request: a request went out with whatever key the agent was constructed with until it 401'd. Now _finalize_routing starts the keepalive (idempotent, process-wide, daemon) whenever an agent resolves onto provider "nous", and prepare_iteration calls _adopt_nous_key_before_expiry(): the agent key is a JWT, its exp is read locally, and inside a 180 s skew the store is re-read under the auth-store lock with force_refresh=False, so the keepalive's (or a peer's) fresh key is adopted without a POST; when none exists, ONE refresh runs there instead of N reactive ones after N 401s. _try_refresh_nous_client_credentials no longer rebuilds the client when the store returns the key already in hand. Live A/B (local server: 401s any bearer but FRESH; store patched to hold FRESH; 40 agents holding a JWT that expires in 30 s fire concurrently): main 40 x 401 then recover, branch 0 x 401. Tests (4): far from expiry the store is not touched; inside the skew the store's fresh key is adopted with force_refresh=False; the same key back from the store is not re-adopted; a real AIAgent routed to nous starts the keepalive and one routed to openrouter does not. --- agent/agent_init.py | 11 +++ agent/client_lifecycle.py | 29 ++++++++ agent/turn_iteration_prep.py | 9 +++ .../test_nous_key_pre_expiry_adoption.py | 74 +++++++++++++++++++ 4 files changed, 123 insertions(+) create mode 100644 tests/agent/test_nous_key_pre_expiry_adoption.py diff --git a/agent/agent_init.py b/agent/agent_init.py index 3421e26df8..ae011ff54e 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -433,6 +433,17 @@ def _finalize_routing(agent, api_mode, credential_pool): with suppress(Exception): agent._get_transport() + # The Nous agent key lives ~1 h. Without the proactive refresher every agent in the process + # discovers expiry reactively, on its own next request, all in the same minute: with 200 + # in-process subagents that was a 401 storm each hour (620 in one run) and the credential + # pool benched the provider for all of them. The gateway and web server start this thread + # at boot; the CLI process (and everything spawned inside it) never did. Idempotent, + # process-wide, daemon. + if agent.provider == "nous": + with suppress(Exception): + from hermes_cli.nous_auth_keepalive import start_nous_auth_keepalive + start_nous_auth_keepalive() + with suppress(Exception): from hermes_cli.model_normalize import ( _AGGREGATOR_PROVIDERS, normalize_model_for_provider diff --git a/agent/client_lifecycle.py b/agent/client_lifecycle.py index 57384bb1c1..8e848a70c2 100644 --- a/agent/client_lifecycle.py +++ b/agent/client_lifecycle.py @@ -3,6 +3,7 @@ shared primary client, single-slot per-request client caches (owner-thread close credential refresh/rotation, route-derived default headers. Extracted from ``run_agent.py``, MRO unchanged.""" import logging import threading +import time from contextlib import suppress from typing import Any, Optional @@ -542,6 +543,8 @@ class ClientLifecycleMixin: api_key, base_url = creds.get("api_key"), creds.get("base_url") if not _valid_credential_pair(api_key, base_url): return False + if str(api_key).strip() == str(self.api_key or "").strip(): + return False # store holds the same key: nothing to adopt, no client rebuild if self.api_mode == "anthropic_messages": self.api_key, self.base_url = api_key.strip(), base_url.strip().rstrip("/") self._anthropic_api_key, self._anthropic_base_url = self.api_key, self.base_url @@ -551,6 +554,32 @@ class ClientLifecycleMixin: self._client_kwargs.pop("default_headers", None) return self._adopt_openai_credentials(api_key, base_url, reason="nous_credential_refresh") + # Adopt a fresh key this many seconds before the one in hand expires. Wider than the store's + # own refresh skew (120 s) so the keepalive has normally already minted the replacement. + _NOUS_KEY_ADOPT_SKEW_S = 180 + + def _adopt_nous_key_before_expiry(self) -> bool: + """Swap in a fresh Nous agent key BEFORE the one in hand expires, so the request never 401s. + + The agent key is a JWT; its ``exp`` is read locally (no network). Inside the skew the store + is re-read under the auth-store lock: the keepalive thread normally holds a fresh key already + (adopt, no POST), otherwise ONE refresh runs and every peer adopts its result. Before this, + every agent in a process learned about the hourly expiry from its own 401, all in the same + minute (620 in one 200-subagent run), and the pool benched the sole credential for all of + them. Returns True when a new key was adopted. + """ + if getattr(self, "provider", "") != "nous" or not getattr(self, "api_key", None): + return False + try: + from hermes_cli.auth_constants import _decode_jwt_claims + exp = _decode_jwt_claims(self.api_key).get("exp") + except Exception: + return False + if not isinstance(exp, (int, float)) or exp - time.time() > self._NOUS_KEY_ADOPT_SKEW_S: + return False + return self._try_refresh_nous_client_credentials(force=False) + + def _resolve_env_credentials(self) -> Optional[tuple]: """Current ``.env``-sourced ``(api_key, base_url, default_base)`` for this provider, or ``None``. diff --git a/agent/turn_iteration_prep.py b/agent/turn_iteration_prep.py index 1b4917758c..d287abb59b 100644 --- a/agent/turn_iteration_prep.py +++ b/agent/turn_iteration_prep.py @@ -50,6 +50,15 @@ def prepare_iteration(agent: Any,*, messages: Any, api_call_count: Any) -> Itera if agent._skill_nudge_interval > 0 and "skill_manage" in agent.valid_tool_names: agent._iters_since_skill += 1 + # Nous agent keys live ~1 h and a single turn can run for hours: adopt the keepalive's fresh + # key before the one in hand expires (local JWT exp read; no network unless inside the skew) + # instead of letting this iteration's request 401. With many agents sharing the hour that + # 401 was a storm, and the pool benched the sole credential for all of them. + try: + agent._adopt_nous_key_before_expiry() + except Exception: + logger.debug("Nous key pre-expiry adoption failed", exc_info=True) + # Drain a /steer sent during the last API call into the newest tool message so # it lands THIS iteration. Never put in a user message (breaks alternation). _pre_api_steer = agent._drain_pending_steer() diff --git a/tests/agent/test_nous_key_pre_expiry_adoption.py b/tests/agent/test_nous_key_pre_expiry_adoption.py new file mode 100644 index 0000000000..320d2e31ca --- /dev/null +++ b/tests/agent/test_nous_key_pre_expiry_adoption.py @@ -0,0 +1,74 @@ +"""Nous agent keys live ~1 h. Every agent in a process must not learn about expiry from its own 401. + +In a 1,393-subagent run the hourly expiry produced 620 authentication_error 401s (177 in one hour) and +the credential pool benched the sole credential for every worker at once; the CLI process never +started the proactive keepalive, and nothing adopted a fresh key before a request was sent. +""" +import base64 +import json +import time +from unittest.mock import patch + +from agent.client_lifecycle import ClientLifecycleMixin + + +def _jwt(exp: float) -> str: + def b64(o): + return base64.urlsafe_b64encode(json.dumps(o).encode()).rstrip(b"=").decode() + return f"{b64({'alg': 'none'})}.{b64({'exp': exp})}.sig" + + +class _Agent(ClientLifecycleMixin): + def __init__(self, key): + self.provider, self.api_mode, self.api_key, self.base_url = "nous", "chat_completions", key, "https://inference-api.nousresearch.com/v1" + self._client_kwargs, self.adopted = {}, [] + + def _adopt_openai_credentials(self, api_key, base_url, *, reason): + self.adopted.append((api_key, reason)) + self.api_key = api_key + return True + + +def test_key_far_from_expiry_is_left_alone_without_touching_the_store(): + agent = _Agent(_jwt(time.time() + 3000)) + with patch("hermes_cli.auth.resolve_nous_runtime_credentials", side_effect=AssertionError("must not hit the store")): + assert agent._adopt_nous_key_before_expiry() is False + assert agent.adopted == [] + + +def test_key_inside_the_skew_adopts_the_stores_fresh_key_without_forcing_a_refresh(): + agent = _Agent(_jwt(time.time() + 60)) + calls = [] + + def resolve(**kw): + calls.append(kw) + return {"api_key": "fresh-key", "base_url": agent.base_url} + + with patch("hermes_cli.auth.resolve_nous_runtime_credentials", side_effect=resolve): + assert agent._adopt_nous_key_before_expiry() is True + assert calls[0]["force_refresh"] is False # the keepalive/peer refresh is adopted, never re-minted + assert agent.adopted == [("fresh-key", "nous_credential_refresh")] + + +def test_same_key_back_from_the_store_is_not_readopted(): + """No client rebuild when the store still holds the key in hand (refresh pending elsewhere).""" + key = _jwt(time.time() + 60) + agent = _Agent(key) + with patch("hermes_cli.auth.resolve_nous_runtime_credentials", return_value={"api_key": key, "base_url": agent.base_url}): + assert agent._adopt_nous_key_before_expiry() is False + assert agent.adopted == [] + + +def test_keepalive_thread_starts_when_an_agent_routes_to_nous(monkeypatch, tmp_path): + """Real construction path: the CLI process builds agents through AIAgent, never through the gateway boot.""" + from run_agent import AIAgent + + monkeypatch.setenv("HERMES_HOME", str(tmp_path / "hh")) + started = [] + monkeypatch.setattr("hermes_cli.nous_auth_keepalive.start_nous_auth_keepalive", lambda: started.append(1)) + AIAgent(api_key="k", base_url="https://inference-api.nousresearch.com/v1", provider="nous", + model="anthropic/claude-fable-5.1", quiet_mode=True, skip_context_files=True, skip_memory=True) + assert started == [1] + AIAgent(api_key="k", base_url="https://openrouter.ai/api/v1", provider="openrouter", + model="anthropic/claude-fable-5.1", quiet_mode=True, skip_context_files=True, skip_memory=True) + assert started == [1] From 55d61f16bab79826d5cced66ee604c8df8037ecd Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 01:33:27 -0700 Subject: [PATCH 007/276] fix(terminal): a foreground timeout above the cap runs as a tracked background process instead of being refused `Foreground timeout Ns exceeds the maximum of 600s` was the second most frequent tool-layer error in the 1,393-agent refactor run: 454 refusals (285 asked 900 s, 41 asked 620, 20 asked 1800), 251 of them test-suite invocations. Every retry was mechanical: lower the timeout (167), split into sleeps (82) or re-send with background=true (86). The schema invited it ("set high for long tasks", max 600) while the suites take 10-30 min. An over-cap foreground timeout is a bounded job the caller wants to wait for. terminal_tool now runs it as a tracked background process with notify_on_complete=true and says so in the result (`promoted_from_foreground`, naming the requested and cap seconds, with "do NOT re-run it"); the schema text describes the new behaviour. The `&`/nohup/setsid and long-lived-server guidance stays a refusal: those need the command itself rewritten, which the tool cannot do safely. Live (real local backend): main refuses `sleep 1; echo LIVE_OK` at timeout=900; branch returns a proc_* session with notify_on_complete and the note, and the command runs once. Tests: the promotion test runs a real command through the real registry and asserts the result shape plus that it executed exactly once in the background; `&` still refuses; schema text updated. Terminal/process suites (27 files) 378 passed. --- .../test_terminal_foreground_timeout_cap.py | 46 ++++++++++------ tools/terminal_tool.py | 52 +++++++++++++++---- 2 files changed, 73 insertions(+), 25 deletions(-) diff --git a/tests/tools/test_terminal_foreground_timeout_cap.py b/tests/tools/test_terminal_foreground_timeout_cap.py index 0081de88a5..7d437c304b 100644 --- a/tests/tools/test_terminal_foreground_timeout_cap.py +++ b/tests/tools/test_terminal_foreground_timeout_cap.py @@ -1,7 +1,8 @@ """Tests for foreground timeout cap in terminal_tool. -Ensures that foreground commands with timeout > FOREGROUND_MAX_TIMEOUT -are rejected with an error suggesting background=true. +A foreground command with timeout > FOREGROUND_MAX_TIMEOUT is promoted to a tracked background +process with notify_on_complete (never refused: in one 1,393-agent run 454 refusals were every one +re-sent lower/split/background, 251 of them test suites). """ import json from unittest.mock import patch, MagicMock @@ -30,22 +31,37 @@ def _make_env_config(**overrides): class TestForegroundTimeoutCap: """FOREGROUND_MAX_TIMEOUT rejects foreground commands that exceed it.""" - def test_foreground_timeout_rejected_above_max(self): - """When model requests timeout > FOREGROUND_MAX_TIMEOUT, return error.""" + def test_foreground_timeout_above_max_is_promoted_to_tracked_background(self, tmp_path, monkeypatch): + """Real local backend, real registry: the command runs (once), the result is a background + session with notify_on_complete and a note naming the requested and cap seconds.""" + import time from tools.terminal_tool import terminal_tool, FOREGROUND_MAX_TIMEOUT + monkeypatch.setenv("HERMES_HOME", str(tmp_path / "hh")) + marker = tmp_path / "ran" + with patch("tools.terminal_tool._get_env_config", return_value=_make_env_config(cwd=str(tmp_path))), \ + patch("tools.terminal_tool._start_cleanup_thread"), \ + patch("tools.terminal_tool._check_all_guards", return_value={"approved": True}): + result = json.loads(terminal_tool(command=f"echo x >> {marker}", timeout=9999)) + + assert result.get("error") is None + assert result["output"] == "Background process started" and result["session_id"].startswith("proc_") + assert result["notify_on_complete"] is True + assert "9999" in result["promoted_from_foreground"] + assert str(FOREGROUND_MAX_TIMEOUT) in result["promoted_from_foreground"] + deadline = time.time() + 10 + while not marker.exists() and time.time() < deadline: + time.sleep(0.05) + assert marker.read_text().count("x") == 1 # ran exactly once, in the background + + def test_shell_backgrounding_is_still_refused(self): + """`&`/nohup need the command rewritten; the tool cannot do that safely, so it still refuses.""" + from tools.terminal_tool import terminal_tool + with patch("tools.terminal_tool._get_env_config", return_value=_make_env_config()), \ patch("tools.terminal_tool._start_cleanup_thread"): - - result = json.loads(terminal_tool( - command="echo hello", - timeout=9999, # Way above max - )) - - assert "error" in result - assert "9999" in result["error"] - assert str(FOREGROUND_MAX_TIMEOUT) in result["error"] - assert "background=true" in result["error"] + result = json.loads(terminal_tool(command="sleep 5 &")) + assert "'&' backgrounding" in result["error"] def test_zero_timeout_rejected(self): """timeout=0 must be rejected, not silently coerced to the default.""" @@ -153,4 +169,4 @@ class TestForegroundMaxTimeoutConstant: from tools.terminal_tool import TERMINAL_SCHEMA, FOREGROUND_MAX_TIMEOUT timeout_desc = TERMINAL_SCHEMA["parameters"]["properties"]["timeout"]["description"] assert str(FOREGROUND_MAX_TIMEOUT) in timeout_desc - assert "background=true" in timeout_desc + assert "background process" in timeout_desc diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index c5e8419822..6815ad542a 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -873,6 +873,17 @@ class _ExecPlan: cwd: str host_cwd: Optional[str] effective_timeout: int + # Set when a foreground call asked for more than FOREGROUND_MAX_TIMEOUT and was promoted to a + # tracked background process instead of being refused (the requested seconds, for the note). + promoted_from_foreground_timeout: Optional[int] = None + + +_PROMOTED_NOTE = ( + "Requested foreground timeout {requested}s exceeds the {cap}s cap, so this command was started as a " + "tracked background process with notify_on_complete=true instead of being refused. Do NOT re-run it. " + "Its completion (exit code + output tail) arrives as a notification; poll with " + "process(action=\"poll\", session_id=...) if you need it sooner." +) def _plan_execution( @@ -936,23 +947,39 @@ def _plan_execution( # value is truthy and would fire an immediate "-Ns" timeout. if timeout is not None and timeout <= 0: raise _Rejected(tool_error(f"timeout must be a positive number of seconds (got {timeout}).")) + promoted = None if not background: + # An over-cap foreground timeout is a bounded job the caller wants to wait for (test suites, + # builds). Refusing it only bought a mechanical retry: 454 refusals in one run, every one + # re-sent lower/split/background. Promote to a tracked background process instead; the + # caller is told in the result. The `&`/nohup/server guidance below stays a refusal: those + # need the command itself rewritten, which the tool cannot do safely. if timeout and timeout > FOREGROUND_MAX_TIMEOUT: - raise _Rejected(tool_error( - f"Foreground timeout {timeout}s exceeds the maximum of " - f"{FOREGROUND_MAX_TIMEOUT}s. Use background=true with " - f"notify_on_complete=true for long-running commands." - )) - guidance = _foreground_background_guidance(command) - if guidance: - raise _Rejected(_error_json(guidance, status="error")) + promoted = timeout + else: + guidance = _foreground_background_guidance(command) + if guidance: + raise _Rejected(_error_json(guidance, status="error")) return _ExecPlan( config=config, env_type=env_type, effective_task_id=effective_task_id, image=image, cwd=cwd, host_cwd=host_cwd, effective_timeout=timeout or config["timeout"], + promoted_from_foreground_timeout=promoted, ) +def _with_promoted_note(result_json: str, requested_timeout: int) -> str: + """Attach the foreground->background promotion note to a spawn result (unchanged on error).""" + try: + data = json.loads(result_json) + except (TypeError, ValueError): + return result_json + if not isinstance(data, dict) or data.get("error"): + return result_json + data["promoted_from_foreground"] = _PROMOTED_NOTE.format(requested=requested_timeout, cap=FOREGROUND_MAX_TIMEOUT) + return json.dumps(data, ensure_ascii=False) + + def _acquire_env(plan: _ExecPlan, task_id: Optional[str]) -> Any: """Cached env for the task, else create it under the per-task creation lock. @@ -1153,14 +1180,19 @@ def terminal_tool( verdict = _run_approval_guards(command, env_type, plan.config, force=force) pty_disabled = pty and _command_requires_pipe_stdin(command) + if plan.promoted_from_foreground_timeout is not None: + background, notify_on_complete, watch_patterns = True, True, None if background: - return spawn_background_process( + result = spawn_background_process( command=command, env=env, env_type=env_type, effective_task_id=effective_task_id, task_id=task_id, session_key=session_key, workdir=workdir, cwd=cwd, effective_pty=pty and not pty_disabled, notify_on_complete=notify_on_complete, watch_patterns=watch_patterns, approval_note=verdict.note, pty_disabled_reason=_PTY_DISABLED_REASON if pty_disabled else None, ) + if plan.promoted_from_foreground_timeout is not None: + result = _with_promoted_note(result, plan.promoted_from_foreground_timeout) + return result return _run_foreground( command, env, plan, task_id=task_id, session_id=session_id, session_key=session_key, @@ -1204,7 +1236,7 @@ TERMINAL_SCHEMA = { }, "timeout": { "type": "integer", - "description": f"Max seconds to wait (default: 180, foreground max: {FOREGROUND_MAX_TIMEOUT}). Returns INSTANTLY when command finishes — set high for long tasks, you won't wait unnecessarily. Foreground timeout above {FOREGROUND_MAX_TIMEOUT}s is rejected; use background=true for longer commands.", + "description": f"Max seconds to wait (default: 180, foreground max: {FOREGROUND_MAX_TIMEOUT}). Returns INSTANTLY when command finishes — set high for long tasks, you won't wait unnecessarily. A foreground timeout above {FOREGROUND_MAX_TIMEOUT}s runs the command as a tracked background process with notify_on_complete=true instead (the result says so; do not re-run it).", "minimum": 1 }, "workdir": { From 6767c06d347e5aac97dd64b5dfd8a534fb92d895 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 01:55:22 -0700 Subject: [PATCH 008/276] feat(delegation): a failed child of a still-running detached fan-out is surfaced to the parent immediately MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Batches join on the slowest sibling before ONE consolidated block re-enters (the design: one results block per fan-out). Failure is the case that should not wait. In the 1,393-agent refactor run every wave-1 child died in the 08:29 401 storm; the parent learned of it at 09:36, when the batch's "unknown outcome" block finally arrived: 66 minutes of a dead wave with nothing running, the single largest idle gap of the run. _run_children_parallel, for DETACHED batches only (the sync path prints a completion line the parent is already watching), pushes ONE type="async_delegation" event with task_failure_notice=True and a single-entry results list when a child ends in a failure status while siblings are still pending. Same event shape and routing fields as the batch result, so every drain/ownership/format path treats it identically; the batch record is not finalized and its consolidated result still arrives unchanged. The formatter renders "[ASYNC DELEGATION TASK FAILED — , task i/n]" with the task, status, error and live transcript path, and says the batch result is still coming. The TUI dedup key distinguishes a notice from the final result and from a sibling's notice. Live through the real detached dispatch (3 children, one fails at 0.1 s, two succeed at 8 s): notice drained at t+0.3 s, batch final at t+8.4 s. On main the parent hears nothing until t+8.4 s. Tests (2): the notice reaches the queue with the record's routing fields while the record stays running, and formats as an early warning; no notice for a finished batch; dedup key differs from the final result. Delegation/registry/notification suites (129 files) 1,344 passed. --- .../test_async_batch_task_failure_notice.py | 48 +++++++++++++++++++ tools/async_delegation.py | 35 ++++++++++++++ tools/delegate_tool_dispatch.py | 16 +++++-- tools/process_registry_notifications.py | 21 ++++++++ tui_gateway/session_notifications.py | 4 ++ 5 files changed, 120 insertions(+), 4 deletions(-) create mode 100644 tests/tools/test_async_batch_task_failure_notice.py diff --git a/tests/tools/test_async_batch_task_failure_notice.py b/tests/tools/test_async_batch_task_failure_notice.py new file mode 100644 index 0000000000..e1b9cce896 --- /dev/null +++ b/tests/tools/test_async_batch_task_failure_notice.py @@ -0,0 +1,48 @@ +"""A failed child of a still-running detached fan-out is surfaced to the parent immediately. + +Batches join on the slowest sibling before ONE consolidated block re-enters. In a 1,393-agent run every wave-1 +child died in a 401 storm at 08:29 and the parent learned of it at 09:36 from the batch's "unknown outcome" +block: 66 minutes of a dead wave with nothing running. +""" +import queue +from unittest.mock import patch + +from tools import async_delegation as ad +from tools.process_registry_notifications import format_process_notification +from tui_gateway.session_notifications import _notification_event_dedup_key + + +def _record(status="running"): + return {"delegation_id": "deleg_x", "status": status, "is_batch": True, "goals": ["a", "b", "c"], "goal": "a", + "session_key": "sk", "origin_ui_session_id": "ui", "origin_session_id": "", "parent_session_id": "root", + "dispatched_at": 1.0, "role": "leaf", "model": "m", "context": None, "toolsets": None} + + +def test_failure_notice_reaches_the_queue_while_the_batch_keeps_running_and_formats_as_early_warning(): + q = queue.Queue() + entry = {"task_index": 1, "status": "error", "error": "401 authentication_error: key invalid", "duration_seconds": 12.5, + "live_transcript": "/tmp/live/task-1.log"} + with patch.object(ad, "_records", {"deleg_x": _record()}), \ + patch("tools.process_registry.process_registry") as reg: + reg.completion_queue = q + ad.push_task_failure_notice("deleg_x", entry, n_tasks=3) + assert ad._records["deleg_x"]["status"] == "running" # not finalized + evt = q.get_nowait() + assert evt["type"] == "async_delegation" and evt["task_failure_notice"] is True + assert (evt["session_key"], evt["origin_ui_session_id"], evt["parent_session_id"]) == ("sk", "ui", "root") + text = format_process_notification(evt) + assert text.startswith("[ASYNC DELEGATION TASK FAILED — deleg_x, task 2/3]") + assert "Task: b" in text and "401 authentication_error" in text and "/tmp/live/task-1.log" in text + assert "consolidated results will still arrive" in text + + +def test_notice_is_not_sent_for_a_finished_batch_and_does_not_dedup_against_the_final_result(): + q = queue.Queue() + with patch.object(ad, "_records", {"deleg_x": _record(status="completed")}), \ + patch("tools.process_registry.process_registry") as reg: + reg.completion_queue = q + ad.push_task_failure_notice("deleg_x", {"task_index": 0, "status": "error"}, n_tasks=3) + assert q.empty() + notice = {"type": "async_delegation", "delegation_id": "deleg_x", "task_failure_notice": True, "results": [{"task_index": 2}]} + final = {"type": "async_delegation", "delegation_id": "deleg_x", "is_batch": True, "results": []} + assert _notification_event_dedup_key(notice) != _notification_event_dedup_key(final) diff --git a/tools/async_delegation.py b/tools/async_delegation.py index 1ef7fbdc27..6fba6071da 100644 --- a/tools/async_delegation.py +++ b/tools/async_delegation.py @@ -678,6 +678,41 @@ def _push_completion_event(record: Dict[str, Any], result: Dict[str, Any], statu "result lost: %s", record.get("delegation_id"), exc) +def push_task_failure_notice(delegation_id: str, entry: Dict[str, Any], *, n_tasks: int) -> None: + """Surface ONE failed child of a still-running detached batch to the parent now, instead of + when the slowest sibling finishes. In a 1,393-agent run every wave-1 child died in a 401 storm + at 08:29 and the parent learned of it at 09:36, when the batch's "unknown outcome" block finally + arrived: 66 minutes of a dead wave with nothing running. The notice rides the same + ``type="async_delegation"`` event shape as the batch result (so every drain/route/format path + treats it identically) with ``task_failure_notice=True`` and a single-entry ``results`` list; the + batch record is NOT finalized and its consolidated result still arrives as before.""" + with _records_lock: + record = _records.get(delegation_id) + if record is None or record.get("status") not in _ACTIVE_STATES: + return + snapshot = dict(record) + try: + from tools.process_registry import process_registry + except Exception as exc: # pragma: no cover + logger.error("Async delegation batch %s: task failure notice dropped (process_registry import): %s", delegation_id, exc) + return + evt = { + "type": "async_delegation", "task_failure_notice": True, "is_batch": True, "n_tasks": n_tasks, + "delegation_id": delegation_id, "results": [entry], + "session_key": snapshot.get("session_key", ""), + "origin_ui_session_id": snapshot.get("origin_ui_session_id", ""), + "origin_session_id": snapshot.get("origin_session_id", ""), + "parent_session_id": snapshot.get("parent_session_id"), + "goal": snapshot.get("goal", ""), "goals": snapshot.get("goals"), "context": snapshot.get("context"), + "toolsets": snapshot.get("toolsets"), "role": snapshot.get("role"), "model": snapshot.get("model"), + "status": "running", "dispatched_at": snapshot.get("dispatched_at") or time.time(), "completed_at": time.time(), + **{k: snapshot[k] for k in _ROUTING_KEYS if snapshot.get(k)}} + try: + process_registry.completion_queue.put(evt) + except Exception as exc: # pragma: no cover + logger.error("Async delegation batch %s: failed to enqueue task failure notice: %s", delegation_id, exc) + + # ── Stale monitor ─────────────────────────────────────────────────────────── def _ensure_stale_monitor() -> None: """Start (once) the stale-delegation monitor thread. One daemon thread serves diff --git a/tools/delegate_tool_dispatch.py b/tools/delegate_tool_dispatch.py index 5f883a2b8f..e6f89170c2 100644 --- a/tools/delegate_tool_dispatch.py +++ b/tools/delegate_tool_dispatch.py @@ -89,7 +89,7 @@ def _report_child_done(parent_agent, spinner_ref, entry, tag, task_labels, n_tas with _quiet("Spinner update_text failed: %s"): spinner_ref.update_text(f"🔀 {'[' + tag + '] ' if tag else ''}{remaining} task{'s' if remaining != 1 else ''} remaining") -def _run_children_parallel(batch: _Batch, results: list, *, honor_parent_interrupt: bool) -> None: +def _run_children_parallel(batch: _Batch, results: list, *, honor_parent_interrupt: bool, detached: bool = False) -> None: """Run the batch's children in parallel, appending entries to ``results`` (sorted by task_index on return, one completion line printed per child). Polls futures with a short ``wait()`` timeout instead of ``as_completed()`` so a wedged child cannot block the parent forever after an interrupt; on parent interrupt the still-pending @@ -126,9 +126,17 @@ def _run_children_parallel(batch: _Batch, results: list, *, honor_parent_interru entry = _entry_of(future, futures[future]) results.append(entry) _report_child_done(parent_agent, spinner_ref, entry, _tag, task_labels, n_tasks, n_tasks - len(results)) + if detached and entry.get("status") in SUBAGENT_FAILURE_STATUSES and len(results) < n_tasks: + # The parent is not watching a spinner: tell it NOW, not when the last sibling finishes. + with _quiet("task failure notice failed", exc_info=True): + from tools.async_delegation import push_task_failure_notice + _i = entry.get("task_index", -1) + _live = batch.live_paths[_i] if isinstance(_i, int) and 0 <= _i < len(batch.live_paths) else None + push_task_failure_notice( + batch.live_deleg_id, {**entry, **({"live_transcript": _live} if _live else {})}, n_tasks=n_tasks) results.sort(key=lambda r: r["task_index"]) # match input order -def _execute_and_aggregate(batch: _Batch, *, honor_parent_interrupt: bool = True) -> dict: +def _execute_and_aggregate(batch: _Batch, *, honor_parent_interrupt: bool = True, detached: bool = False) -> dict: """Run all built children, join, finalize (hooks + cost rollup), return the combined dict. Shared by the sync path and the background runner: even in the background the batch JOINS on itself here so ONE consolidated results block re-enters the conversation. Live transcripts are finalized but retained as the full-fidelity record @@ -138,7 +146,7 @@ def _execute_and_aggregate(batch: _Batch, *, honor_parent_interrupt: bool = True if len(batch.task_list) == 1: results.append(batch.run_child(*batch.children[0])) else: - _run_children_parallel(batch, results, honor_parent_interrupt=honor_parent_interrupt) + _run_children_parallel(batch, results, honor_parent_interrupt=honor_parent_interrupt, detached=detached) _finalize_child_results(results, batch.task_list, batch.children, batch.parent_agent) total_duration = round(time.monotonic() - batch.overall_start, 2) @@ -319,7 +327,7 @@ def _dispatch_background(batch: _Batch) -> str: role=batch.top_role, model=batch.creds["model"], session_key=session_key, origin_ui_session_id=origin_ui_session_id, origin_session_id=wake_sid, parent_session_id=getattr(parent_agent, "session_id", None), - runner=lambda: _execute_and_aggregate(batch, honor_parent_interrupt=False), + runner=lambda: _execute_and_aggregate(batch, honor_parent_interrupt=False, detached=True), interrupt_fn=_batch_interrupt, max_async_children=_get_max_async_children(), # Reuse the live-transcript directory's id (when created) so the returned delegation_id matches # cache/delegation/live//. diff --git a/tools/process_registry_notifications.py b/tools/process_registry_notifications.py index 15c9355867..1821ea0199 100644 --- a/tools/process_registry_notifications.py +++ b/tools/process_registry_notifications.py @@ -113,6 +113,25 @@ def _preamble(evt: dict, title: str, intro: str, completed_at: float, *, with_go return lines +def _format_task_failure_notice(evt: dict, deleg_id: str) -> str: + """One child of a still-running fan-out failed: say which, why, and that the batch goes on.""" + (r,) = (evt.get("results") or [{}])[:1] or [{}] + goals, idx, n = evt.get("goals") or [], r.get("task_index", 0), evt.get("n_tasks") or len(evt.get("goals") or []) + goal = goals[idx] if idx < len(goals) else r.get("goal", "") + err = str(r.get("error") or "").strip().replace("\n", " ")[:400] + lines = [ + f"[ASYNC DELEGATION TASK FAILED — {deleg_id}, task {idx + 1}/{n}]", + "One subagent in a background fan-out you dispatched has failed while its siblings are still running. " + "The batch's consolidated results will still arrive when the last sibling finishes; this is an early " + "warning so you can re-dispatch or investigate now instead of then.", + f"Task: {goal}" if goal else "", + f"Status: {r.get('status', '?')} Duration: {r.get('duration_seconds', '?')}s" + (f"\nError: {err}" if err else ""), + ] + if r.get("live_transcript"): + lines.append(f"Live transcript: {r['live_transcript']}") + return "\n".join(line for line in lines if line) + + def _format_batch_delegation(evt: dict, deleg_id: str, completed_at: float) -> str: """Consolidated block for a delegate_task fan-out that finished as one unit.""" results, goals = evt.get("results") or [], evt.get("goals") or [] @@ -163,6 +182,8 @@ def _format_async_delegation(evt: dict) -> str: and result, so an agent deep in unrelated context can act on it or re-dispatch.""" deleg_id = evt.get("delegation_id", "unknown") completed_at = evt.get("completed_at") or time.time() + if evt.get("task_failure_notice"): + return _format_task_failure_notice(evt, deleg_id) if evt.get("is_batch") or isinstance(evt.get("results"), list): return _format_batch_delegation(evt, deleg_id, completed_at) status, summary, error = evt.get("status") or "completed", evt.get("summary"), evt.get("error") diff --git a/tui_gateway/session_notifications.py b/tui_gateway/session_notifications.py index 3457bfa7ed..e7f6887d42 100644 --- a/tui_gateway/session_notifications.py +++ b/tui_gateway/session_notifications.py @@ -104,6 +104,10 @@ def _notification_event_dedup_key(evt: dict) -> tuple: evt_type = evt.get("type", "completion") if evt_type == "async_delegation": # No process session_id: else every completion keys as ("", "async_delegation") and the second is suppressed forever. + # An early per-task failure notice must not collapse with the batch's final result (nor with a sibling's notice). + if evt.get("task_failure_notice"): + task_idx = ((evt.get("results") or [{}])[0] or {}).get("task_index", "") + return (evt.get("delegation_id", ""), evt_type, "task_failure", task_idx) return (evt.get("delegation_id", ""), evt_type) extra = _DEDUP_EXTRA_FIELDS.get("watch_overflow_" if evt_type.startswith("watch_overflow_") else evt_type, ()) return (evt.get("session_id", ""), evt_type, *(evt.get(f, 0 if f == "suppressed" else "") for f in extra)) From a1ffb27ef8417cd5a870d31d1b984e5ba562d2d7 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 02:01:04 -0700 Subject: [PATCH 009/276] feat(file_tools): write_file says when it just re-sent a large file that was already on disk In the 1,393-agent refactor run write_file carried 92.8M chars of content (23.2M output tokens, ~$580, 20% of all output tokens). 661 of those calls were read -> whole-file rewrite of an existing file >20k chars (24.9M chars, ~$155): 130k-char agent/model_metadata.py, 109k model_setup_flows.py, 107k google_chat/adapter.py, re-sent in full to change a few lines. Meanwhile patch succeeded 4,602/4,617 (99.7%) at 1.3k chars per call. The tool result is the only place the model reads, so the cost signal goes there. When an overwrite targets an existing file and both old and new content are >=20k chars, write_file diffs them by line (difflib, autojunk off) and, if >=80% of lines are unchanged, adds a `hint` naming the unchanged count, the changed count, the size, and "use patch". The write itself is unchanged and still happens; new files, small files and genuine rewrites get no hint. 300 ms on the 119k-char model_metadata.py. Tests (2): a 40k-char file with 2 lines changed gets the hint and is written; a new file, a small file and a wholly rewritten file get none. --- tests/tools/test_write_file_rewrite_hint.py | 34 +++++++++++++++++++ tools/file_tools.py | 37 +++++++++++++++++++++ 2 files changed, 71 insertions(+) create mode 100644 tests/tools/test_write_file_rewrite_hint.py diff --git a/tests/tools/test_write_file_rewrite_hint.py b/tests/tools/test_write_file_rewrite_hint.py new file mode 100644 index 0000000000..5ed14c3b60 --- /dev/null +++ b/tests/tools/test_write_file_rewrite_hint.py @@ -0,0 +1,34 @@ +"""write_file tells the caller when it just re-sent a large file that was already on disk. + +In one 1,393-agent run 661 read->whole-file-rewrites of >20k-char files cost ~25M output chars (~$155) +while `patch` (1.3k chars/call) succeeded 99.7% of the time; the tool result is the only place to say so. +""" +import json + +from tools.file_tools import write_file_tool + + +def _big(n_lines=800): + return "\n".join(f"line {i}: " + "x" * 40 for i in range(n_lines)) + "\n" # ~40k chars + + +def test_rewrite_of_a_large_file_with_few_changes_gets_a_patch_hint(tmp_path): + f = tmp_path / "mod.py" + old = _big() + f.write_text(old, encoding="utf-8") + new = old.replace("line 400:", "line 400 (edited):").replace("line 401:", "line 401 (edited):") + r = json.loads(write_file_tool(str(f), new, task_id="t")) + assert r.get("error") is None and f.read_text(encoding="utf-8") == new # the write still happens + assert "use patch" in r["hint"] and "798 of 800 lines were already on disk" in r["hint"] + + +def test_new_files_small_files_and_real_rewrites_get_no_hint(tmp_path): + new_file = tmp_path / "new.py" + assert "hint" not in json.loads(write_file_tool(str(new_file), _big(), task_id="t")) + small = tmp_path / "small.py" + small.write_text("a\nb\n", encoding="utf-8") + assert "hint" not in json.loads(write_file_tool(str(small), "a\nc\n", task_id="t")) + big = tmp_path / "big.py" + big.write_text(_big(), encoding="utf-8") + rewritten = "\n".join(f"other {i}: " + "y" * 40 for i in range(800)) + "\n" + assert "hint" not in json.loads(write_file_tool(str(big), rewritten, task_id="t")) diff --git a/tools/file_tools.py b/tools/file_tools.py index 280f896fcc..ac1c06e22c 100644 --- a/tools/file_tools.py +++ b/tools/file_tools.py @@ -709,6 +709,40 @@ def _note_edited(task_id: str, paths: list[str], path_to_resolved: dict, session file_state.note_write(task_id, path_to_resolved[p]) +# Whole-file rewrite hint: an overwrite of an existing file this large whose new content keeps at least +# this fraction of the old lines is a patch written the expensive way. In one 1,393-agent run 661 such +# rewrites of >20k-char files cost ~25M output chars (~$155) where `patch` averaged 1.3k chars/call. +_REWRITE_HINT_MIN_CHARS = 20_000 +_REWRITE_HINT_MIN_UNCHANGED = 0.80 + + +def _whole_file_rewrite_hint(resolved: str | None, new_content: str) -> str | None: + """Return a hint when ``new_content`` mostly re-sends what is already on disk at ``resolved``.""" + if not resolved or len(new_content) < _REWRITE_HINT_MIN_CHARS: + return None + try: + old = Path(resolved).read_text(encoding="utf-8", errors="replace") + except OSError: + return None + if len(old) < _REWRITE_HINT_MIN_CHARS: + return None + old_lines, new_lines = old.splitlines(), new_content.splitlines() + if not old_lines: + return None + import difflib + matcher = difflib.SequenceMatcher(None, old_lines, new_lines, autojunk=False) + unchanged = sum(size for _, _, size in matcher.get_matching_blocks()) + ratio = unchanged / max(len(old_lines), len(new_lines)) + if ratio < _REWRITE_HINT_MIN_UNCHANGED: + return None + changed = max(len(old_lines), len(new_lines)) - unchanged + return ( + f"{unchanged:,} of {len(new_lines):,} lines were already on disk ({ratio:.0%} unchanged); ~{changed:,} " + f"line(s) actually changed. Re-sending a {len(new_content):,}-char file costs output tokens for every " + "unchanged line; for edits like this use patch (old_string/new_string), which sends only the changed region." + ) + + def write_file_tool(path: str, content: str, task_id: str = "default", cross_profile: bool = False, session_id: str | None = None) -> str: @@ -741,9 +775,12 @@ def write_file_tool(path: str, content: str, task_id: str = "default", # subagents; different paths stay fully parallel. _lock.enter_context(file_state.lock_path(_resolved)) warnings = _edit_warnings([path], path_to_resolved, task_id) + rewrite_hint = _whole_file_rewrite_hint(_resolved, content) result_dict = _get_file_ops(task_id).write_file(_resolved or path, content).to_dict() if warnings: result_dict["_warning"] = warnings[0] + if rewrite_hint and not result_dict.get("error"): + result_dict["hint"] = rewrite_hint if _resolved: # Always report the ABSOLUTE path written so a wrong-cwd mismatch # is visible in the response instead of silently landing elsewhere. From 7e2dbcfb33416fc38c957a422343bea600bd1f6a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 02:03:50 -0700 Subject: [PATCH 010/276] fix(goal): /goal kicks the loop with a short pointer when the user's last message already carries the goal `/goal ` queues the goal text as the next user turn to start the loop. When that text is what the user just pasted (a handoff note, a plan the agent already has), the kickoff re-sends it verbatim: in the 1,393-agent run `/goal <2,000-char handoff>` came 16 minutes after the same note was pasted as a message, and the agent spent 11 API calls / 6 min deciding it was a replay, with the note duplicated in context from then on. _goal_kick_prompt() compares the normalized goal against the LAST user message (string or block content); when contained, the kick is a one-line '[Goal set] Continue with the goal you were just given; there is no need to re-read it.' Otherwise the goal text is sent as before. Both /goal and /goal draft use it. Tests (2): a just-pasted goal (plain and block content) kicks with the pointer; a new goal, an unrelated last message, and a goal pasted in an OLDER turn kick verbatim. --- hermes_cli/cli_commands_mixin.py | 25 +++++++++++++++-- tests/cli/test_cli_goal_kick_prompt.py | 37 ++++++++++++++++++++++++++ 2 files changed, 60 insertions(+), 2 deletions(-) create mode 100644 tests/cli/test_cli_goal_kick_prompt.py diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index 84fe41b0a0..7f58ad33cd 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -2223,6 +2223,27 @@ class CLICommandsMixin: else: self._goal_set(mgr, arg) + # `/goal ` kicks the loop by queueing the goal text as the next user turn. When that text + # is what the user JUST said (a pasted handoff note, a plan the agent already has), re-sending + # it makes the agent spend a turn deciding it is a replay (11 API calls, 6 min, in one run) and + # duplicates ~2k tokens of context. A short pointer starts the loop just as well. + _GOAL_ALREADY_SEEN_KICK = "[Goal set] Continue with the goal you were just given; there is no need to re-read it." + + def _goal_kick_prompt(self, goal: str) -> str: + """The goal text, or a short pointer when the last user message already carries it.""" + last_user = "" + for msg in reversed(getattr(self, "conversation_history", None) or []): + if msg.get("role") == "user": + content = msg.get("content") + if isinstance(content, list): + content = " ".join(str(b.get("text", "")) for b in content if isinstance(b, dict)) + last_user = str(content or "") + break + goal_norm = " ".join(goal.split()) + if goal_norm and goal_norm in " ".join(last_user.split()): + return self._GOAL_ALREADY_SEEN_KICK + return goal + def _kick_goal(self, prompt: str) -> bool: """Queue ``prompt`` as the next turn so the loop starts without a separate message.""" try: @@ -2310,7 +2331,7 @@ class CLICommandsMixin: _cp(_dim_line(f"After each turn, a judge model checks if the goal is done{against}. " "Hermes keeps working until it is, you pause/clear it, or the budget is " "exhausted. Use /goal status, /goal show, /goal pause, /goal resume, /goal clear.")) - self._kick_goal(state.goal) + self._kick_goal(self._goal_kick_prompt(state.goal)) def _print_goal_set(self, state, contract_label: str) -> None: _cp(f" ⊙ Goal set ({state.max_turns}-turn budget): {state.goal}") @@ -2343,7 +2364,7 @@ class CLICommandsMixin: else: _cp(_dim_line("Couldn't draft a contract (aux model unavailable) — running as a " "free-form goal. The per-turn judge still applies.")) - self._kick_goal(state.goal) + self._kick_goal(self._goal_kick_prompt(state.goal)) def _handle_loop_command(self, cmd: str) -> None: """Dispatch /loop — recurring in-session wakeups: ``/loop [interval] [--times N] diff --git a/tests/cli/test_cli_goal_kick_prompt.py b/tests/cli/test_cli_goal_kick_prompt.py new file mode 100644 index 0000000000..ae57abc900 --- /dev/null +++ b/tests/cli/test_cli_goal_kick_prompt.py @@ -0,0 +1,37 @@ +"""`/goal ` kicks the loop with a short pointer when the user's last message already carries the goal. + +In one run `/goal <2,000-char handoff note>` was issued 16 minutes after the same note had been pasted as +a user message; the kickoff re-sent it and the agent spent 11 API calls / 6 min deciding it was a replay. +""" +import queue + +from hermes_cli.cli_commands_mixin import CLICommandsMixin + + +def _cli(history): + cli = CLICommandsMixin.__new__(CLICommandsMixin) + cli.conversation_history = history + cli._pending_input = queue.Queue() + return cli + + +HANDOFF = "HANDOFF: resume round 3 integration.\n - merge r3-16 first\n - then run the full suite" + + +def test_goal_that_the_user_just_pasted_kicks_with_a_pointer_not_the_text(): + cli = _cli([{"role": "user", "content": "Here is the plan.\n\n" + HANDOFF + "\n\nGo."}, + {"role": "assistant", "content": "ok"}]) + assert cli._goal_kick_prompt(HANDOFF) == CLICommandsMixin._GOAL_ALREADY_SEEN_KICK + # block-style content is handled too + cli = _cli([{"role": "user", "content": [{"type": "text", "text": HANDOFF}]}]) + assert cli._goal_kick_prompt(HANDOFF) == CLICommandsMixin._GOAL_ALREADY_SEEN_KICK + + +def test_a_new_goal_or_a_goal_from_an_older_turn_is_kicked_verbatim(): + assert _cli([])._goal_kick_prompt("Ship the release") == "Ship the release" + cli = _cli([{"role": "user", "content": "Something unrelated"}]) + assert cli._goal_kick_prompt(HANDOFF) == HANDOFF + # only the LAST user message counts: the agent has moved on since an older paste + cli = _cli([{"role": "user", "content": HANDOFF}, {"role": "assistant", "content": "done"}, + {"role": "user", "content": "now something else"}]) + assert cli._goal_kick_prompt(HANDOFF) == HANDOFF From 41b88ba68a2abc0c3986fe2c4178817e78c847f0 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:25:05 +0530 Subject: [PATCH 011/276] chore: map contributor email for salvaged #62023 (sagitario-jpn) --- .../emails/301893261+sagitario-jpn@users.noreply.github.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/301893261+sagitario-jpn@users.noreply.github.com diff --git a/contributors/emails/301893261+sagitario-jpn@users.noreply.github.com b/contributors/emails/301893261+sagitario-jpn@users.noreply.github.com new file mode 100644 index 0000000000..f7019c373c --- /dev/null +++ b/contributors/emails/301893261+sagitario-jpn@users.noreply.github.com @@ -0,0 +1,2 @@ +sagitario-jpn +# PR #62023 salvage From 7cad5da0f06b7efebcc9eae6bbe2d0109c525af7 Mon Sep 17 00:00:00 2001 From: Sagitario JPN <301893261+sagitario-jpn@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:25:05 +0530 Subject: [PATCH 012/276] fix(tools): every sandbox creator uses the shared container_config shaper (#76906, #87995, #84027, #100019) The prompt backend-probe and execute_code each kept a private (key, default) table for container_config and drifted from the terminal tool's: the probe omitted docker_network, so under docker_network: false it started a bridge-networked container; execute_code omitted docker_extra_args, docker_forward_env and docker_env, so a sandbox created from that path lost the operator's settings. Both now call terminal_tool_backends._container_config_from_config. Salvaged from #62023 by @sagitario-jpn (the probe docker_network fix), widened to the whole class. --- agent/prompt_builder.py | 14 ++++---------- tools/code_execution_tool.py | 13 ++++--------- 2 files changed, 8 insertions(+), 19 deletions(-) diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index 9e00dd0dc2..7ee3cde17a 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -845,13 +845,6 @@ def _tenv_read(name: str, default: str = "") -> str: _BACKEND_IMAGE_KEYS = {b: f"{b}_image" for b in ("docker", "singularity", "modal", "daytona")} # (config key, default) pairs forwarded to _create_environment's container_config. -_CONTAINER_CONFIG_DEFAULTS = ( - ("container_cpu", 1), ("container_memory", 5120), ("container_disk", 51200), ("container_persistent", True), - ("modal_mode", "auto"), ("docker_volumes", []), ("docker_mount_cwd_to_workspace", False), - ("docker_forward_env", []), ("docker_env", {}), ("docker_run_as_host_user", False), ("docker_extra_args", []), - ("docker_shm_size", "1g"), ("docker_persist_across_processes", True), ("docker_shared_container_key", ""), - ("docker_orphan_reaper", True), -) # Single-line POSIX probe; `2>/dev/null` keeps a missing binary from polluting output. _BACKEND_PROBE_CMD = ( "printf 'os=%s\\nkernel=%s\\nhome=%s\\ncwd=%s\\nuser=%s\\n' \"$(uname -s 2>/dev/null || echo unknown)\" " @@ -862,16 +855,17 @@ _BACKEND_PROBE_CMD = ( def _run_backend_probe(env_type: str, terminal_tool) -> str: """Execute the probe command inside a freshly built backend; "" when it yields nothing.""" - from tools.terminal_tool_backends import _create_environment, _ssh_config_from_config + from tools.terminal_tool_backends import _container_config_from_config, _create_environment, _ssh_config_from_config from tools.terminal_tool_lifecycle import _cleanup_env config = terminal_tool._get_env_config() - # Mirrors tools/terminal_tool.py's live-command assembly (`_create_environment` is the factory). + # Same container_config shaper as the live terminal path: a private copy of the key table here + # drifted (no docker_network) and gave the probe a bridge-networked container under lockdown. env = _create_environment( env_type=env_type, image=config.get(_BACKEND_IMAGE_KEYS[env_type], "") if env_type in _BACKEND_IMAGE_KEYS else "", cwd=config.get("cwd", ""), timeout=config.get("timeout", 180), ssh_config=_ssh_config_from_config(config) if env_type == "ssh" else None, - container_config=({k: config.get(k, d) for k, d in _CONTAINER_CONFIG_DEFAULTS} + container_config=(_container_config_from_config(config) if terminal_tool._is_container_backend(env_type) else None), task_id="prompt-backend-probe", host_cwd=config.get("host_cwd"), ) diff --git a/tools/code_execution_tool.py b/tools/code_execution_tool.py index bf56c3e7c2..a06bc98679 100644 --- a/tools/code_execution_tool.py +++ b/tools/code_execution_tool.py @@ -394,17 +394,10 @@ def _call(tool_name, args): # ---- Remote execution support (file-based RPC via terminal backend) ---- -# execute_code's container_config keys (a subset of terminal_tool's; the create path fills the rest). -_CONTAINER_CONFIG_DEFAULTS = ( - ("container_cpu", 1), ("container_memory", 5120), ("container_disk", 51200), ("container_persistent", True), - ("vercel_runtime", ""), ("docker_volumes", []), ("docker_run_as_host_user", False), ("docker_network", True), -) - - def _get_or_create_env(task_id: str): """``(env, env_type)`` — the environment the terminal/file tools share for *task_id*, created on first use (same double-checked per-task lock pattern as file_tools._get_file_ops).""" - from tools.terminal_tool_backends import _create_environment, _ssh_config_from_config + from tools.terminal_tool_backends import _container_config_from_config, _create_environment, _ssh_config_from_config from tools.terminal_tool import ( _active_environments, _env_lock, _get_env_config, _last_activity, _start_cleanup_thread, _creation_locks, _creation_locks_lock, _task_env_overrides, @@ -431,7 +424,9 @@ def _get_or_create_env(task_id: str): overrides = _task_env_overrides.get(effective_task_id, {}) container_config = None if _is_container_backend(env_type): - container_config = {key: config.get(key, default) for key, default in _CONTAINER_CONFIG_DEFAULTS} + # Shared shaper: execute_code's own key subset dropped docker_extra_args / docker_forward_env / + # docker_env, so a sandbox created from this path lost the operator's configured settings. + container_config = _container_config_from_config(config) logger.info("Creating new %s environment for execute_code task %s...", env_type, effective_task_id[:8]) env = _create_environment( From deaebd9b1713745e35164f8673bd2d4c38d99e5f Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:25:05 +0530 Subject: [PATCH 013/276] test(tools): probe and execute_code hand _create_environment the same container_config as the terminal tool Replaces the source-reading AST check with a behaviour contract. Red on origin/main. --- tests/tools/test_docker_network_config.py | 75 ++++++++++++----------- 1 file changed, 39 insertions(+), 36 deletions(-) diff --git a/tests/tools/test_docker_network_config.py b/tests/tools/test_docker_network_config.py index 73c48e988d..5af4af9098 100644 --- a/tests/tools/test_docker_network_config.py +++ b/tests/tools/test_docker_network_config.py @@ -18,45 +18,48 @@ def test_terminal_env_config_reads_docker_network_toggle(monkeypatch): assert config["docker_network"] is False -def test_sibling_container_config_sites_carry_docker_network(): - """Every container_config dict that carries docker_run_as_host_user must - also carry docker_network — otherwise that code path silently falls back - to networked containers while the terminal path honors the lockdown - (the probe/exec asymmetry reported on issue #46358). - """ - import ast - import inspect - +def test_every_sandbox_creator_passes_the_full_container_config(monkeypatch): + """The terminal tool, execute_code and the prompt backend-probe must hand ``_create_environment`` + the SAME container_config keys. Each used to keep a private (key, default) table and drifted: + the probe lost ``docker_network`` (bridge-networked probe under lockdown, #46358/#76906/#87995), + execute_code lost ``docker_extra_args``/``docker_forward_env``/``docker_env`` (#84027/#100019).""" + import agent.prompt_builder as prompt_builder import tools.code_execution_tool as code_execution_tool - import tools.file_tools as file_tools + import tools.terminal_tool_backends as backends - # file_tools no longer builds its own container_config: it goes through - # the shared _create_configured_env, which is what carries docker_network. - assert "_create_configured_env(" in inspect.getsource(file_tools) + config = {"env_type": "docker", "cwd": "/root", "timeout": 60, "docker_network": False, + "docker_extra_args": ["--user", "1009:1009"], "docker_forward_env": ["DATABASE_URL"], + "docker_env": {"FOO": "bar"}, "docker_image": "debian:bookworm-slim"} + expected = backends._container_config_from_config(config) + seen: list = [] - for module in (terminal_tool, code_execution_tool): - tree = ast.parse(inspect.getsource(module)) - sites = 0 - for node in ast.walk(tree): - # Accept a dict literal or a (key, default) pair table that a dict - # comprehension is built from. - if isinstance(node, ast.Dict): - keys = {k.value for k in node.keys if isinstance(k, ast.Constant)} - elif isinstance(node, ast.Tuple) and node.elts and all( - isinstance(e, ast.Tuple) and len(e.elts) == 2 and isinstance(e.elts[0], ast.Constant) - for e in node.elts - ): - keys = {e.elts[0].value for e in node.elts} - else: - continue - if "docker_run_as_host_user" in keys: - sites += 1 - assert "docker_network" in keys, ( - f"{module.__name__} builds a container_config with " - f"docker_run_as_host_user but without docker_network " - f"(line {node.lineno})" - ) - assert sites >= 1, f"expected at least one container_config site in {module.__name__}" + class _Env: + cwd = "/root" + + def execute(self, *a, **k): + return {"output": "", "returncode": 0} + + def cleanup(self, **k): + pass + + def _fake_create(**kwargs): + seen.append(kwargs["container_config"]) + return _Env() + + # Both creators late-import their collaborators from terminal_tool / terminal_tool_backends. + monkeypatch.setattr(backends, "_create_environment", _fake_create) + monkeypatch.setattr(terminal_tool, "_get_env_config", lambda: config) + monkeypatch.setattr(terminal_tool, "_select_image", lambda *a, **k: "img") + monkeypatch.setattr(terminal_tool, "_resolve_task_host_cwd", lambda *a, **k: None) + monkeypatch.setattr(terminal_tool, "_start_cleanup_thread", lambda: None) + monkeypatch.setattr(terminal_tool, "_task_env_overrides", {}) + monkeypatch.setattr(terminal_tool, "_active_environments", {}) + + prompt_builder._run_backend_probe("docker", terminal_tool) + code_execution_tool._get_or_create_env("cc-task") + + assert seen == [expected, expected] # probe, then execute_code + assert expected["docker_network"] is False and expected["docker_extra_args"] == ["--user", "1009:1009"] def _reuse_guard_harness( From 98bf8b20734ae6372047a800eb4e4d3dc5862406 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:07:21 -0700 Subject: [PATCH 014/276] fix(approval): a quoted $(...)/backtick body keeps its command boundaries when newlines are masked; a backtick operand is not a closer Independent review of the first fix found two defects. Bounding the grep lexer removed the old "malformed payload" floor, and that floor had been the ONLY thing blocking a hardline command hidden after a newline inside a double-quoted $(grep ...) substitution: _mask_quoted_newlines treated the newline as quoted DATA and replaced it with a space, so the later command-start pass saw `/dev/null reboot` as an operand and the public guards approved a hardline command with no callback. A substitution body inside double quotes is executable; the masker now recurses into $(...) and backtick bodies with a fresh quote state so a newline there stays a command boundary. The witness now blocks as "system shutdown/reboot" (the correct verdict), as do the ; / && / | / nested / backtick variants. Second: the lexer stopped at ANY backtick as if it closed an enclosing substitution, so a grep with a backtick OPERAND lost its operands and was reported malformed (a new false positive). A backtick opened after the grep is an operand substitution; only an unmatched one closes the enclosing command. Tests: the six hardline-inside-substitution shapes block as themselves; the public guard blocks with zero approval callbacks; the backtick operand, a data pattern and a quoted-newline commit message stay allowed. Approval/hardline suites: 637 passed; the one failure (real-binary sort payload probe) fails identically on main. --- .../test_approval_grep_in_substitution.py | 34 +++++++++++++ tools/approval_detection.py | 48 +++++++++++++++++-- 2 files changed, 77 insertions(+), 5 deletions(-) diff --git a/tests/tools/test_approval_grep_in_substitution.py b/tests/tools/test_approval_grep_in_substitution.py index abfff1359b..5cd85b3d1b 100644 --- a/tests/tools/test_approval_grep_in_substitution.py +++ b/tests/tools/test_approval_grep_in_substitution.py @@ -39,3 +39,37 @@ def test_genuinely_unbalanced_quoting_still_fails_closed(): def test_hardline_patterns_unchanged(): for command in ("rm -rf /", 'echo "$(rm -rf /)"', "sudo shutdown -h now"): assert detect_hardline_command(command)[0] is True, command + + +class TestExecutableSubstitutionBodiesStayExecutable: + """Bounding the grep lexer removed the old 'malformed' floor; the newline masker must then treat a + substitution body inside double quotes as CODE, or a newline-separated hardline command hides as an + operand (independent review witness: blocked on main as 'malformed', approved on the first fix).""" + + @pytest.mark.parametrize("cmd", [ + 'echo "$(grep -P \'safe\' /dev/null\nreboot)"', + 'echo "$(grep x f; reboot)"', + 'echo "$(grep x f && reboot)"', + 'echo "$(grep x f | shutdown -h now)"', + 'echo "$(echo $(grep x f)\nreboot)"', + 'echo "`grep x f\nreboot`"', + ]) + def test_hardline_command_inside_a_quoted_substitution_blocks_as_itself(self, cmd): + blocked, reason = detect_hardline_command(cmd) + assert blocked and reason == "system shutdown/reboot" + + def test_public_guard_blocks_without_offering_approval(self, tmp_path, monkeypatch): + monkeypatch.setenv("HERMES_HOME", str(tmp_path / "hh")) + from tools.approval import check_dangerous_command + calls = [] + result = check_dangerous_command('echo "$(grep -P \'safe\' /dev/null\nreboot)"', "local", + approval_callback=lambda *a, **k: calls.append(a) or False) + assert result["approved"] is False and "hardline" in result["message"].lower() and calls == [] + + @pytest.mark.parametrize("cmd", [ + "grep -e `echo needle` file", # backtick OPERAND (opened after the grep) is not a closer + "grep -F 'sudo reboot' notes.md", # data pattern + 'git commit -m "fix\nsudo reboot handling"', # quoted data newline + ]) + def test_benign_shapes_stay_allowed(self, cmd): + assert detect_hardline_command(cmd) == (False, None) diff --git a/tools/approval_detection.py b/tools/approval_detection.py index 23b7613abd..9790b30a15 100644 --- a/tools/approval_detection.py +++ b/tools/approval_detection.py @@ -578,6 +578,14 @@ def _command_parser_limit_exceeded(command: str) -> bool: return sum(command.count(char) for char in ";&|\n") >= _MAX_DETECTION_SEGMENTS +def _backtick_end_from(segment: str, i: int) -> int | None: + """Index of the backtick closing the one opened at ``i``, or None when none follows.""" + j = segment.find("`", i + 1) + while j != -1 and segment[j - 1] == "\\": + j = segment.find("`", j + 1) + return None if j == -1 else j + + def _shell_tokens_with_spans(segment: str, start: int): """Return shell words as ``(value, start, end, quoted)`` or ``None`` on malformed quoting. Deliberately small lexer that never expands shell syntax; it exists to keep source spans (which @@ -591,6 +599,9 @@ def _shell_tokens_with_spans(segment: str, start: int): is the canonical shape).""" tokens, value, token_start, quote = [], [], None, None depth = 0 # $(...) nesting opened AFTER start; a closer at depth 0 ends the enclosing substitution + # A backtick opened AFTER start is an operand substitution (``grep -e `cmd` f``); the matching + # closer belongs to it, not to an enclosing backtick the command might sit inside. + in_backtick = False def flush(end: int) -> None: raw = segment[token_start:end] @@ -610,7 +621,15 @@ def _shell_tokens_with_spans(segment: str, start: int): continue if segment.startswith("$(", i): depth += 1 - elif ch == ")" or ch == "`": + elif ch == "`": + if in_backtick: + in_backtick = False + elif depth == 0 and _backtick_end_from(segment, i) is None: + end_at = i # unmatched: it closes the substitution this command sits inside + break + else: + in_backtick = True + elif ch == ")": if depth == 0: end_at = i break @@ -1042,10 +1061,29 @@ def _mask_quoted_newlines(command: str) -> str: exactly as the shell would, so masking them cannot hide a runnable command.""" if "\n" not in command: return command - return "".join( - " " if quote and kind == "char" and command[i] == "\n" else command[i:j] - for kind, i, j, quote in _scan_shell(command) - ) + return _mask_quoted_newlines_span(command, 0, len(command)) + + +def _mask_quoted_newlines_span(command: str, start: int, end: int) -> str: + """``_mask_quoted_newlines`` over ``command[start:end]``. A ``$(...)`` / backtick substitution + inside double quotes is EXECUTABLE, not data: its body is re-scanned with a fresh quote state so a + newline that separates commands inside it survives as a command boundary. Masking it as quoted + data turned ``"$(grep x f\nreboot)"`` into ``... f reboot)``, which no later stage can tell from an + operand; the pre-fix scanner only caught it by refusing the whole command as malformed.""" + out: list[str] = [] + for kind, i, j, quote in _scan_shell(command, start, end, subst="q"): + if kind == "subst": + # j is the index just past the closer; keep the opener and closer, recurse into the body. + body_start = i + (2 if command.startswith("$(", i) else 1) + body_end = j - 1 + out.append(command[i:body_start]) + out.append(_mask_quoted_newlines_span(command, body_start, body_end)) + out.append(command[body_end:j]) + elif quote and kind == "char" and command[i] == "\n": + out.append(" ") + else: + out.append(command[i:j]) + return "".join(out) def _iter_shell_command_word_spans(command: str): From ef897bfd7ec37e77b41991fa0c57140c4d1f1d51 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:09:24 -0700 Subject: [PATCH 015/276] fix(nous): pre-expiry adoption only ever swaps to a key for the SAME account Independent review of the first fix found a credential-identity takeover: _adopt_nous_key_before_expiry() called the singleton resolver unconditionally, so an agent running on an explicitly supplied (or pool-selected) account-A key near expiry was moved onto the logged-in account-B key from auth.json before the A key had even failed, and the next real SDK request went out as B. That silently changes who is billed. The adoption now reads the `sub` claim of the key in hand and passes it as require_account; _try_refresh_nous_client_credentials refuses any replacement whose `sub` differs (logged at INFO, current key kept). A key with no `sub` is not adopted proactively at all. The reactive 401 path is unchanged, and same-account adoption (the keepalive's or a peer's fresh key) still works. Verified with the reviewer's own live probe (real AIAgent -> real prepare_iteration -> real auth.json transaction -> real OpenAI SDK -> loopback capture): explicit-account case account-A -> account-A, adopted_fresh=False (was A -> B); same-account still adopts; 12 concurrent near-expiry agents still 0 x 401. Tests (2 new): a fresh key for a different account is never adopted; a key without an account claim is left alone without touching the store. --- agent/client_lifecycle.py | 27 ++++++++++++++++--- .../test_nous_key_pre_expiry_adoption.py | 26 +++++++++++++++--- 2 files changed, 45 insertions(+), 8 deletions(-) diff --git a/agent/client_lifecycle.py b/agent/client_lifecycle.py index 8e848a70c2..f617870920 100644 --- a/agent/client_lifecycle.py +++ b/agent/client_lifecycle.py @@ -525,7 +525,7 @@ class ClientLifecycleMixin: return False return self._adopt_openai_credentials(api_key, base_url, reason=f"{self.provider}_credential_refresh") - def _try_refresh_nous_client_credentials(self, *, force: bool = True) -> bool: + def _try_refresh_nous_client_credentials(self, *, force: bool = True, require_account: str | None = None) -> bool: # Portal serves anthropic/* on the native Messages route, so either client kind may hold the expiring JWT. if self.provider != "nous" or self.api_mode not in ("chat_completions", "anthropic_messages"): return False @@ -545,6 +545,18 @@ class ClientLifecycleMixin: return False if str(api_key).strip() == str(self.api_key or "").strip(): return False # store holds the same key: nothing to adopt, no client rebuild + if require_account is not None: + try: + from hermes_cli.auth_constants import _decode_jwt_claims + new_account = _decode_jwt_claims(str(api_key)).get("sub") + except Exception: + new_account = None + if str(new_account or "") != require_account: + logger.info( + "Nous pre-expiry adoption skipped: the store's key belongs to a different account " + "than the one in hand; keeping the current credential." + ) + return False if self.api_mode == "anthropic_messages": self.api_key, self.base_url = api_key.strip(), base_url.strip().rstrip("/") self._anthropic_api_key, self._anthropic_base_url = self.api_key, self.base_url @@ -567,17 +579,24 @@ class ClientLifecycleMixin: every agent in a process learned about the hourly expiry from its own 401, all in the same minute (620 in one 200-subagent run), and the pool benched the sole credential for all of them. Returns True when a new key was adopted. + + Identity guard: the replacement must belong to the SAME account (``sub`` claim) as the key + in hand. The store holds the logged-in singleton; an agent running on an explicitly supplied + or pool-selected key for a different account must never be silently moved onto it (that + changes who is billed). When either side lacks a ``sub`` nothing is adopted here; the + reactive 401 path is unchanged. """ if getattr(self, "provider", "") != "nous" or not getattr(self, "api_key", None): return False try: from hermes_cli.auth_constants import _decode_jwt_claims - exp = _decode_jwt_claims(self.api_key).get("exp") + claims = _decode_jwt_claims(self.api_key) except Exception: return False - if not isinstance(exp, (int, float)) or exp - time.time() > self._NOUS_KEY_ADOPT_SKEW_S: + exp, account = claims.get("exp"), claims.get("sub") + if not account or not isinstance(exp, (int, float)) or exp - time.time() > self._NOUS_KEY_ADOPT_SKEW_S: return False - return self._try_refresh_nous_client_credentials(force=False) + return self._try_refresh_nous_client_credentials(force=False, require_account=str(account)) def _resolve_env_credentials(self) -> Optional[tuple]: diff --git a/tests/agent/test_nous_key_pre_expiry_adoption.py b/tests/agent/test_nous_key_pre_expiry_adoption.py index 320d2e31ca..82be617ee7 100644 --- a/tests/agent/test_nous_key_pre_expiry_adoption.py +++ b/tests/agent/test_nous_key_pre_expiry_adoption.py @@ -12,10 +12,10 @@ from unittest.mock import patch from agent.client_lifecycle import ClientLifecycleMixin -def _jwt(exp: float) -> str: +def _jwt(exp: float, sub: str = "acct-A") -> str: def b64(o): return base64.urlsafe_b64encode(json.dumps(o).encode()).rstrip(b"=").decode() - return f"{b64({'alg': 'none'})}.{b64({'exp': exp})}.sig" + return f"{b64({'alg': 'none'})}.{b64({'exp': exp, 'sub': sub})}.sig" class _Agent(ClientLifecycleMixin): @@ -40,14 +40,32 @@ def test_key_inside_the_skew_adopts_the_stores_fresh_key_without_forcing_a_refre agent = _Agent(_jwt(time.time() + 60)) calls = [] + fresh = _jwt(time.time() + 3600) # same account A + def resolve(**kw): calls.append(kw) - return {"api_key": "fresh-key", "base_url": agent.base_url} + return {"api_key": fresh, "base_url": agent.base_url} with patch("hermes_cli.auth.resolve_nous_runtime_credentials", side_effect=resolve): assert agent._adopt_nous_key_before_expiry() is True assert calls[0]["force_refresh"] is False # the keepalive/peer refresh is adopted, never re-minted - assert agent.adopted == [("fresh-key", "nous_credential_refresh")] + assert agent.adopted == [(fresh, "nous_credential_refresh")] + + +def test_a_fresh_key_for_a_different_account_is_never_adopted(): + """Independent-review witness: an explicitly supplied account-A key near expiry was replaced by the + logged-in singleton's account-B key and the next real request went out as B. Identity is preserved.""" + agent = _Agent(_jwt(time.time() + 60, sub="acct-A")) + other = _jwt(time.time() + 3600, sub="acct-B") + with patch("hermes_cli.auth.resolve_nous_runtime_credentials", return_value={"api_key": other, "base_url": agent.base_url}): + assert agent._adopt_nous_key_before_expiry() is False + assert agent.adopted == [] and agent.api_key != other + + +def test_a_key_without_an_account_claim_is_left_alone_proactively(): + agent = _Agent(_jwt(time.time() + 60, sub="")) + with patch("hermes_cli.auth.resolve_nous_runtime_credentials", side_effect=AssertionError("must not hit the store")): + assert agent._adopt_nous_key_before_expiry() is False def test_same_key_back_from_the_store_is_not_readopted(): From 70214e9ab8eef74234e168f6d92c678f2c369ec4 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:12:37 -0700 Subject: [PATCH 016/276] fix(delegation): an interim task-failure notice never claims, acknowledges or dedups against the batch's final result Independent review of the first version reproduced a final-result loss with real dispatch, SQLite claims and the delivery consumers: the interim notice carried the batch's delegation_id and every consumer claims and acknowledges durable rows by that id alone. A busy parent that drained the notice first acknowledged the FINAL result's row; the consolidated result could then never be claimed, and would not replay after restart. The gateway's in-memory dedup keyed on (type, delegation_id) and suppressed the final (and every sibling notice) the same way. The notice is now recognised as a non-durable event at both boundaries: claim_event_delivery returns the empty token for it (is_interim_delegation_event), the gateway preflight does not claim the durable row for it, and the gateway dedup identity carries the task index so notices are distinct from the final and from each other (the TUI key already did). Reviewer's own probe re-run on this head: gateway receives ['notice', 'notice', 'final'] (was ['notice']); busy-parent case: the notice takes no claim token, the final's claim succeeds and its row stays pending until delivered (was: final claim None, row 'delivered' with an undelivered result inside). Tests (2 new): the notice is non-durable at the claim boundary; the gateway identity separates two notices and the final into three. --- gateway/run_notifications.py | 13 ++++++++++-- .../test_async_batch_task_failure_notice.py | 21 +++++++++++++++++++ tools/async_delegation.py | 11 +++++++++- 3 files changed, 42 insertions(+), 3 deletions(-) diff --git a/gateway/run_notifications.py b/gateway/run_notifications.py index fc356f608c..c5ce3cb4b2 100644 --- a/gateway/run_notifications.py +++ b/gateway/run_notifications.py @@ -950,7 +950,14 @@ class GatewayNotificationsMixin: evt_type = str(evt.get("type") or "") if evt_type == "async_delegation": producer_id = str(evt.get("delegation_id") or "") - return (evt_type, producer_id, "") if producer_id else None + if not producer_id: + return None + if evt.get("task_failure_notice"): + # An interim per-task notice is its own producer event: it must not mark the + # batch's final result as already delivered, nor a sibling's notice. + task_idx = ((evt.get("results") or [{}])[0] or {}).get("task_index", "") + return (evt_type, producer_id, f"task_failure:{task_idx}") + return (evt_type, producer_id, "") if evt_type == "completion": producer_id = str(evt.get("session_id") or "") started_at = evt.get("started_at") @@ -1035,7 +1042,9 @@ class GatewayNotificationsMixin: """ claim = self._CompletionClaim() evt_type = evt.get("type") - if evt_type == "async_delegation": + # An interim per-task notice shares the batch's delegation_id but is not the durable + # completion; claiming that row here would acknowledge the FINAL result before it exists. + if evt_type == "async_delegation" and not evt.get("task_failure_notice"): claim.delegation_id = str(evt.get("delegation_id") or "") if claim.delegation_id: try: diff --git a/tests/tools/test_async_batch_task_failure_notice.py b/tests/tools/test_async_batch_task_failure_notice.py index e1b9cce896..427855cab3 100644 --- a/tests/tools/test_async_batch_task_failure_notice.py +++ b/tests/tools/test_async_batch_task_failure_notice.py @@ -46,3 +46,24 @@ def test_notice_is_not_sent_for_a_finished_batch_and_does_not_dedup_against_the_ notice = {"type": "async_delegation", "delegation_id": "deleg_x", "task_failure_notice": True, "results": [{"task_index": 2}]} final = {"type": "async_delegation", "delegation_id": "deleg_x", "is_batch": True, "results": []} assert _notification_event_dedup_key(notice) != _notification_event_dedup_key(final) + + +def test_interim_notice_never_claims_or_acknowledges_the_batch_final_row(tmp_path, monkeypatch): + """Independent-review witness: a busy parent that drained the notice first acknowledged the FINAL + result's durable row, and the consolidated result was never delivered (nor replayed after restart).""" + monkeypatch.setenv("HERMES_HOME", str(tmp_path / "hh")) + notice = {"type": "async_delegation", "delegation_id": "deleg_x", "task_failure_notice": True, "results": [{"task_index": 0}]} + final = {"type": "async_delegation", "delegation_id": "deleg_x", "is_batch": True, "results": []} + # The notice is a non-durable event: empty token, no row touched. + assert ad.claim_event_delivery(notice, "tui-poller") == "" + ad.complete_event_delivery(notice, "") + assert ad.is_interim_delegation_event(notice) and not ad.is_interim_delegation_event(final) + + +def test_gateway_dedup_identity_separates_notices_from_the_final_and_from_each_other(): + from gateway.run_notifications import GatewayNotificationsMixin + ident = GatewayNotificationsMixin._completion_delivery_identity + n0 = {"type": "async_delegation", "delegation_id": "d", "task_failure_notice": True, "results": [{"task_index": 0}]} + n1 = {"type": "async_delegation", "delegation_id": "d", "task_failure_notice": True, "results": [{"task_index": 1}]} + final = {"type": "async_delegation", "delegation_id": "d", "is_batch": True, "results": []} + assert len({ident(n0), ident(n1), ident(final)}) == 3 diff --git a/tools/async_delegation.py b/tools/async_delegation.py index 6fba6071da..97a0df13d4 100644 --- a/tools/async_delegation.py +++ b/tools/async_delegation.py @@ -323,8 +323,17 @@ def claim_completion_delivery(delegation_id: str, claim_id: str) -> bool: return cur.rowcount == 1 +def is_interim_delegation_event(evt: Dict[str, Any]) -> bool: + """An early per-task notice for a batch that is still running. It shares the batch's + ``delegation_id`` but is NOT the durable completion: it must never claim, acknowledge or + dedup against the final result's row (independent review reproduced exactly that loss).""" + return evt.get("type") == "async_delegation" and bool(evt.get("task_failure_notice")) + + def claim_event_delivery(evt: Dict[str, Any], consumer: str) -> Optional[str]: - """Claim a durable delegation event; non-durable events need no token.""" + """Claim a durable delegation event; non-durable events (and interim notices) need no token.""" + if is_interim_delegation_event(evt): + return "" delegation_id = str(evt.get("delegation_id") or "") if evt.get("type") == "async_delegation" else "" if not delegation_id: return "" From 2a93bdaeba6cf9295513271de84f327fcd03d060 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:36:24 +0530 Subject: [PATCH 017/276] chore: map contributor email for salvaged #72429 (JonthanaHanh) --- .../emails/92574114+JonthanaHanh@users.noreply.github.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/92574114+JonthanaHanh@users.noreply.github.com diff --git a/contributors/emails/92574114+JonthanaHanh@users.noreply.github.com b/contributors/emails/92574114+JonthanaHanh@users.noreply.github.com new file mode 100644 index 0000000000..2ed6eff33d --- /dev/null +++ b/contributors/emails/92574114+JonthanaHanh@users.noreply.github.com @@ -0,0 +1,2 @@ +JonthanaHanh +# PR #72429 salvage From d78cdd7119cc17c9d268501adbc5e5da2f5c1a72 Mon Sep 17 00:00:00 2001 From: JonthanaHanh <92574114+JonthanaHanh@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:36:24 +0530 Subject: [PATCH 018/276] fix(tools): truncation footers name the path the SANDBOX sees, not the host path (#72389, #81984, #77015) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit web_extract / browser_snapshot / delegate_task spill their full text under HERMES_HOME/cache, which is mounted read-only into docker/modal (at /root/.hermes) and synced under ~/.hermes for ssh/daytona/vercel — but the footer told the agent the HOST path, so read_file inside the sandbox got 'File not found'. Translate through the existing credential_files.to_agent_visible_cache_path (what tool_result_storage already does); local and singularity are unchanged. Salvaged from #72429 by @JonthanaHanh (the web_extract sites), widened to every footer. --- tools/browser_tool_snapshot.py | 4 ++++ tools/delegate_tool_results.py | 4 ++++ tools/web_tools_truncate.py | 5 +++++ 3 files changed, 13 insertions(+) diff --git a/tools/browser_tool_snapshot.py b/tools/browser_tool_snapshot.py index f502441229..c35a512382 100644 --- a/tools/browser_tool_snapshot.py +++ b/tools/browser_tool_snapshot.py @@ -79,6 +79,10 @@ def _truncate_snapshot(snapshot_text: str, max_chars: Optional[int] = None) -> s return snapshot_text stored_path = _store_full_snapshot(snapshot_text) + if stored_path: + # Agent-visible path: read_file runs inside the active backend (#72389). + from tools.credential_files import to_agent_visible_cache_path + stored_path = to_agent_visible_cache_path(stored_path) lines = snapshot_text.split('\n') result: list[str] = [] diff --git a/tools/delegate_tool_results.py b/tools/delegate_tool_results.py index c3b08a58be..3d655d4b50 100644 --- a/tools/delegate_tool_results.py +++ b/tools/delegate_tool_results.py @@ -200,6 +200,10 @@ def _trim_summary_with_footer(summary: str, cap: int, task_index: int) -> tuple[ tail = tail[nl + 1:] spill_path = _spill_summary_to_file(task_index, summary) + if spill_path: + # Agent-visible path: the parent's read_file runs inside the active backend (#81984, #77015). + from tools.credential_files import to_agent_visible_cache_path + spill_path = to_agent_visible_cache_path(spill_path) footer_lines = [ "", "─" * 8 + " [SUMMARY TRUNCATED] " + "─" * 8, f"Showing {len(head):,} chars (head) + {len(tail):,} chars (tail) " diff --git a/tools/web_tools_truncate.py b/tools/web_tools_truncate.py index 4570eb2f7f..1cee30e81f 100644 --- a/tools/web_tools_truncate.py +++ b/tools/web_tools_truncate.py @@ -99,6 +99,11 @@ def _truncate_with_footer(content: str, url: str, char_limit: int) -> tuple[str, tail = tail[nl + 1:] stored_path = _store_full_text(url, content) + if stored_path: + # The footer is read by the AGENT, whose read_file runs inside the active backend: render the + # path where docker/modal/ssh/... see the mounted cache, not the host path (#72389, #81984). + from tools.credential_files import to_agent_visible_cache_path + stored_path = to_agent_visible_cache_path(stored_path) footer_lines = [ "", "─" * 8 + " [TRUNCATED] " + "─" * 8, f"Showing {len(head):,} chars (head) + {len(tail):,} chars (tail) " From ef89a7f1c9ef61c3d84c6471d321a0b1d565f263 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:36:24 +0530 Subject: [PATCH 019/276] test(tools): every truncation footer renders the backend-visible cache path under docker; local keeps the host path Red on origin/main. --- .../test_truncation_footer_sandbox_paths.py | 41 +++++++++++++++++++ 1 file changed, 41 insertions(+) create mode 100644 tests/tools/test_truncation_footer_sandbox_paths.py diff --git a/tests/tools/test_truncation_footer_sandbox_paths.py b/tests/tools/test_truncation_footer_sandbox_paths.py new file mode 100644 index 0000000000..0028edfa0e --- /dev/null +++ b/tests/tools/test_truncation_footer_sandbox_paths.py @@ -0,0 +1,41 @@ +"""Truncation footers tell the AGENT where the full text lives, and the agent's read_file runs inside +the active terminal backend: under docker the mounted cache is at /root/.hermes, so a host path in the +footer is unreadable from the sandbox (#72389, #81984, #77015).""" + +import os +import re + +import tools.browser_tool_snapshot as browser_tool_snapshot +import tools.delegate_tool_results as delegate_tool_results +import tools.web_tools_truncate as web_tools_truncate + + +def _footer_path(text: str) -> str: + """Every truncation footer tells the agent how to page the full text via ``read_file path="..."``.""" + return re.search(r'read_file path="([^"]+)"', text).group(1) + + +def test_truncation_footers_render_the_sandbox_visible_cache_path(tmp_path, monkeypatch): + home = tmp_path / ".hermes" + monkeypatch.setenv("HERMES_HOME", str(home)) + monkeypatch.setenv("TERMINAL_ENV", "docker") + body = "\n".join(f"row {i}" for i in range(5000)) + + web_out, _ = web_tools_truncate._truncate_with_footer(body, "https://example.com/doc", 3000) + snap_out = browser_tool_snapshot._truncate_snapshot(body, 3000) + deleg_out, _ = delegate_tool_results._trim_summary_with_footer(body, 3000, 0) + + for text, subdir in ((web_out, "cache/web"), (snap_out, "cache/web"), (deleg_out, "cache/delegation")): + path = _footer_path(text) + assert path.startswith(f"/root/.hermes/{subdir}/"), path + # The bytes still live on the host under HERMES_HOME; only the rendered path is translated. + assert os.path.exists(str(home / subdir / os.path.basename(path))) + + +def test_local_backend_footer_keeps_the_host_path(tmp_path, monkeypatch): + home = tmp_path / ".hermes" + monkeypatch.setenv("HERMES_HOME", str(home)) + monkeypatch.setenv("TERMINAL_ENV", "local") + body = "\n".join(f"row {i}" for i in range(5000)) + out, _ = web_tools_truncate._truncate_with_footer(body, "https://example.com/doc", 3000) + assert _footer_path(out).startswith(str(home)) From a2ff25a8b143c9125371cbb41cfcaa56b55ddba9 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:40:09 +0530 Subject: [PATCH 020/276] chore: map contributor email for salvaged #97265 (Liuzikaii) --- contributors/emails/2319582736@qq.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/2319582736@qq.com diff --git a/contributors/emails/2319582736@qq.com b/contributors/emails/2319582736@qq.com new file mode 100644 index 0000000000..db2d1cab47 --- /dev/null +++ b/contributors/emails/2319582736@qq.com @@ -0,0 +1,2 @@ +Liuzikaii +# PR #97265 salvage From c99dbba33e509e5faaa5ad23ef67900d02cd44ee Mon Sep 17 00:00:00 2001 From: liuzikaii <2319582736@qq.com> Date: Sat, 5 Sep 2026 18:40:09 +0530 Subject: [PATCH 021/276] fix(code-execution): remote kernels are keyed by their sandbox tool set (#97263) The hermes_tools stub module a remote kernel imports is generated from sandbox_tools once, at spawn, but the registry key was (owner, env_type, task_env_id) only: a later execute_code call with a different tool set (skill loaded, toolset toggled) reused the kernel and got stale stubs. The tool set is now part of the key; a different set gets its own kernel and the over-cap eviction keeps the newest. Salvaged from #97265 by @Liuzikaii. --- tools/code_kernel_remote.py | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/tools/code_kernel_remote.py b/tools/code_kernel_remote.py index 47bd24a878..85c8b36c9e 100644 --- a/tools/code_kernel_remote.py +++ b/tools/code_kernel_remote.py @@ -146,8 +146,10 @@ class RemoteKernel: logger.debug(failure, exc_info=True) -def _kernel_key(owner: str, env_type: str, task_env_id: str) -> Tuple: - return (owner, "remote", env_type, task_env_id) +def _kernel_key(owner: str, env_type: str, task_env_id: str, sandbox_tools: frozenset) -> Tuple: + """The hermes_tools stub module is generated from ``sandbox_tools`` once, at spawn, so a kernel + is only reusable by calls with the SAME tool set; a different set gets its own kernel.""" + return (owner, "remote", env_type, task_env_id, tuple(sorted(sandbox_tools))) # Registry + lock shared-shape with code_kernel; teardown runs outside the lock. @@ -243,7 +245,7 @@ def _acquire_remote_kernel(env, env_type: str, owner: str, task_env_id: str, idle_exit: int) -> Tuple[Optional[RemoteKernel], bool, bool, bool]: """Find/respawn the owner's kernel: (kernel|None, reused, state_reset, state_lost); reaps idle-expired entries on the way in.""" - key = _kernel_key(owner, env_type, task_env_id) + key = _kernel_key(owner, env_type, task_env_id, sandbox_tools) state_lost = state_reset = False with _REGISTRY.lock: expired = _reap_unlocked(idle_exit) @@ -309,7 +311,7 @@ def execute_in_remote_kernel( env, env_type, owner, task_env_id, sandbox_tools, reset=reset, idle_exit=idle_exit) if kernel is None: return None # fail open to per-call - key = _kernel_key(owner, env_type, task_env_id) + key = _kernel_key(owner, env_type, task_env_id, sandbox_tools) kernel.last_used = time.monotonic() with _REGISTRY.lock: kernel.attached += 1 From 7b8cf4c6314fd29d734acf24aa70cf361743a693 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:40:10 +0530 Subject: [PATCH 022/276] test(code-execution): a changed tool set spawns a fresh remote kernel with fresh stubs Red on origin/main. --- tests/tools/test_code_kernel_remote.py | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/tests/tools/test_code_kernel_remote.py b/tests/tools/test_code_kernel_remote.py index bf8470dcc2..f7eabefa0d 100644 --- a/tests/tools/test_code_kernel_remote.py +++ b/tests/tools/test_code_kernel_remote.py @@ -74,10 +74,11 @@ def _cell(status="ok", stdout="", execution_count=1, **kw): return payload -def _run(env, code="print(1)", *, task="t1", reset=False, timeout=10): +def _run(env, code="print(1)", *, task="t1", reset=False, timeout=10, + tools=frozenset({"read_file"})): return execute_in_remote_kernel( code, env=env, env_type="ssh", task_env_id=task, - sandbox_tools=frozenset({"read_file"}), timeout=timeout, + sandbox_tools=tools, timeout=timeout, max_tool_calls=5, reset=reset, ) @@ -176,6 +177,19 @@ class TestDeathDetection(RemoteKernelBase): class TestOwnershipIsolation(RemoteKernelBase): + def test_changed_tool_set_spawns_kernel_with_fresh_stubs(self): + env = ScriptedEnv(_spawn_ok_handlers([_cell(), _cell()])) + _run(env, tools=frozenset({"read_file"})) + _run(env, tools=frozenset({"web_search"})) + + self.assertEqual(len(_REMOTE_KERNELS), 2) + self.assertEqual(sum(1 for c in env.commands if "nohup" in c), 2) + keyed_tool_sets = {key[-1] for key in _REMOTE_KERNELS} + self.assertEqual( + keyed_tool_sets, + {("read_file",), ("web_search",)}, + ) + def test_delegated_children_get_their_own_remote_kernels(self): """Same invariant as local (#94647 review fix): the child context qualifier must key a DIFFERENT remote kernel.""" From 09138852500bc02062008a565942c0c9176b2fc3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:18:41 -0700 Subject: [PATCH 023/276] fix(delegation): summary headroom uses the aggregator's own prompt size; unknown usage means the static ceiling, never zero context Independent review found two holes in the first fix. A parent with no usage row yet was treated as 0 tokens used, so a 190K/200K prompt received a 384K-char dynamic summary budget instead of ~4K; the budget now returns None (static ceiling only) when nothing is known. And under MoA the folded usage includes advisor prompts that are not in the parent's context, over-stating the prompt size and wrongly truncating summaries; turn_usage now records the aggregator's pre-fold prompt_tokens as _last_prompt_size_tokens and the budget reads that first. Tests (2 new): unknown usage -> None; MoA-folded and unfolded parents with the same real prompt get the same budget. --- agent/turn_usage.py | 3 +++ tests/tools/test_delegate_summary_budget.py | 20 ++++++++++++++++++++ tools/delegate_tool_results.py | 21 +++++++++++++++++---- 3 files changed, 40 insertions(+), 4 deletions(-) diff --git a/agent/turn_usage.py b/agent/turn_usage.py index 131e2cd348..517b216ffe 100644 --- a/agent/turn_usage.py +++ b/agent/turn_usage.py @@ -143,6 +143,9 @@ def record_response_usage( # Stash canonical usage for on_turn_complete(); keep the latest call's. agent._last_turn_usage = dict(usage_dict) + # The parent's CURRENT prompt size for headroom math (delegate summary budgets): the + # aggregator's own prompt, never the MoA-folded total (advisor prompts are not in this context). + agent._last_prompt_size_tokens = int(aggregator_usage.prompt_tokens or 0) # Persist only provider-confirmed context lengths, not probe tiers. if getattr(compressor, "_context_probed", False): diff --git a/tests/tools/test_delegate_summary_budget.py b/tests/tools/test_delegate_summary_budget.py index f5b81fed85..6a3de40c17 100644 --- a/tests/tools/test_delegate_summary_budget.py +++ b/tests/tools/test_delegate_summary_budget.py @@ -88,3 +88,23 @@ def test_budget_uses_current_prompt_size_not_the_session_sum(): fresh = _FakeParent(context_length=200_000, used_tokens=30_000, max_tokens=8_000) assert _parent_summary_char_budget(long_lived, 1) == _parent_summary_char_budget(fresh, 1) assert _parent_summary_char_budget(long_lived, 1) > _MIN_SUMMARY_CHARS + + +def test_unknown_parent_usage_means_static_ceiling_not_zero_context(): + """Independent-review witness: a parent with no usage yet was treated as 0 tokens used, so a 190K/200K + prompt got a 384K-char summary budget instead of ~4K.""" + from types import SimpleNamespace + from tools.delegate_tool_results import _parent_summary_char_budget + parent = SimpleNamespace(context_compressor=SimpleNamespace(context_length=200_000, max_tokens=0), + _last_turn_usage=None) + assert _parent_summary_char_budget(parent, 1) is None + + +def test_moa_fold_does_not_inflate_the_parents_prompt_size(): + """MoA folds advisor prompts into reported usage; the parent's context holds only the aggregator's.""" + from types import SimpleNamespace + from tools.delegate_tool_results import _parent_summary_char_budget + cc = SimpleNamespace(context_length=200_000, max_tokens=0) + folded = SimpleNamespace(context_compressor=cc, _last_turn_usage={"prompt_tokens": 190_000}, _last_prompt_size_tokens=50_000) + unfolded = SimpleNamespace(context_compressor=cc, _last_turn_usage={"prompt_tokens": 50_000}) + assert _parent_summary_char_budget(folded, 1) == _parent_summary_char_budget(unfolded, 1) diff --git a/tools/delegate_tool_results.py b/tools/delegate_tool_results.py index f45e7c13ee..a921807c9d 100644 --- a/tools/delegate_tool_results.py +++ b/tools/delegate_tool_results.py @@ -218,6 +218,20 @@ def _trim_summary_with_footer(summary: str, cap: int, task_index: int) -> tuple[ footer_lines.append("─" * 37) return head + "\n\n[... middle omitted — see footer ...]\n\n" + tail + "\n".join(footer_lines), spill_path +def _parent_prompt_size_tokens(parent_agent) -> Optional[int]: + """The parent's current prompt size: the aggregator's own last ``prompt_tokens`` (pre-MoA-fold), else + the last provider usage. ``None`` when no request has completed yet: the caller then applies the + static ceiling only. Treating "no usage yet" as zero handed a 190K/200K parent a 384K-char budget.""" + size = getattr(parent_agent, "_last_prompt_size_tokens", None) + if isinstance(size, (int, float)) and size > 0: + return int(size) + last_usage = getattr(parent_agent, "_last_turn_usage", None) or {} + used = last_usage.get("prompt_tokens") if isinstance(last_usage, dict) else None + if isinstance(used, (int, float)) and used > 0: + return int(used) + return None + + def _parent_summary_char_budget(parent_agent, n_summaries: int) -> Optional[int]: """Per-summary char budget from the parent's *remaining* context headroom (context length − the parent's current prompt size − the compressor's output reserve), a fraction of it split across the batch at ~4 @@ -233,10 +247,9 @@ def _parent_summary_char_budget(parent_agent, n_summaries: int) -> Optional[int] context_length = getattr(compressor, "context_length", None) if not isinstance(context_length, int) or context_length <= 0: return None - last_usage = getattr(parent_agent, "_last_turn_usage", None) or {} - used_tokens = last_usage.get("prompt_tokens") if isinstance(last_usage, dict) else None - if not isinstance(used_tokens, (int, float)) or used_tokens < 0: - used_tokens = 0 + used_tokens = _parent_prompt_size_tokens(parent_agent) + if used_tokens is None: + return None # no usage yet and nothing to estimate from: static ceiling only, never "zero context" headroom_tokens = context_length - int(used_tokens) - int(getattr(compressor, "max_tokens", 0) or 0) if headroom_tokens <= 0: return _MIN_SUMMARY_CHARS # parent already over budget: floor only From 21bcd0ce6c94da5b67a4ceab46dbd51fdcd29340 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 02:36:56 -0700 Subject: [PATCH 024/276] chore(models): OpenRouter and Nous pickers drop the undated deepseek-v4-flash in favor of the 0731 snapshot The bare deepseek/deepseek-v4-flash slug is the pre-snapshot release; both aggregators now carry deepseek-v4-flash-0731 (and Nous exposes the rolling ~deepseek/deepseek-v4-flash-latest alias). Keeping both rows in the curated picker just duplicates the flash tier. - OPENROUTER_MODELS (derives the nous list): drop deepseek/deepseek-v4-flash - model-catalog.json regenerated Deliberately KEPT: alibaba-token-plan / opencode-go / commandcode / deepseek direct plugin fallback_models + default_aux_model (bare id is the wire slug those providers serve), DEFAULT_CONTEXT_LENGTHS / reasoning floor / pricing snapshot entries (manually-typed id still behaves), and the deepseek-chat/deepseek-reasoner -> deepseek-v4-flash alias normalization. --- hermes_cli/models_catalog_static.py | 2 +- website/static/api/model-catalog.json | 9 +-------- 2 files changed, 2 insertions(+), 9 deletions(-) diff --git a/hermes_cli/models_catalog_static.py b/hermes_cli/models_catalog_static.py index 49bfe81b42..97d96b9924 100644 --- a/hermes_cli/models_catalog_static.py +++ b/hermes_cli/models_catalog_static.py @@ -35,7 +35,7 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [ "openai/gpt-5.6-terra", "openai/gpt-5.6-terra-pro", "openai/gpt-5.6-luna", "openai/gpt-5.6-luna-pro", "openai/gpt-5.5", "openai/gpt-5.5-pro", "openai/gpt-5.4-mini", "google/gemini-3.1-pro-preview", "google/gemini-3.8-flash", "google/gemini-3.7-flash", "x-ai/grok-4.6", "deepseek/deepseek-v4-pro", - "deepseek/deepseek-v4-pro-0813", "deepseek/deepseek-v4-flash", "deepseek/deepseek-v4-flash-0731", + "deepseek/deepseek-v4-pro-0813", "deepseek/deepseek-v4-flash-0731", "qwen/qwen3.8-max-0902", "qwen/qwen3.8-flash", "moonshotai/kimi-k3", "minimax/minimax-m3", "z-ai/glm-5.3", "z-ai/glm-5.3-flash", "z-ai/glm-5.2", "xiaomi/mimo-v2.5-pro", "tencent/hy4-preview", "tencent/hy3", "stepfun/step-3.7-flash", "nvidia/nemotron-3-super-120b-a12b", "meta/muse-spark-1.2", diff --git a/website/static/api/model-catalog.json b/website/static/api/model-catalog.json index 2b3a54b2e6..7dd0151138 100644 --- a/website/static/api/model-catalog.json +++ b/website/static/api/model-catalog.json @@ -1,6 +1,6 @@ { "version": 1, - "updated_at": "2026-09-05T06:42:02Z", + "updated_at": "2026-09-05T09:33:43Z", "metadata": { "source": "hermes-agent repo", "docs": "https://hermes-agent.nousresearch.com/docs/reference/model-catalog" @@ -128,10 +128,6 @@ "id": "deepseek/deepseek-v4-pro-0813", "description": "dated snapshot of v4-pro" }, - { - "id": "deepseek/deepseek-v4-flash", - "description": "" - }, { "id": "deepseek/deepseek-v4-flash-0731", "description": "dated snapshot of v4-flash" @@ -334,9 +330,6 @@ { "id": "deepseek/deepseek-v4-pro-0813" }, - { - "id": "deepseek/deepseek-v4-flash" - }, { "id": "deepseek/deepseek-v4-flash-0731" }, From 1f3912d277fbb3fde936d5144cd49f9924a4b171 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:23:55 -0700 Subject: [PATCH 025/276] fix(goal): the TUI judge also sees only its own session's processes; a lifted barrier resumes the loop from the idle hook Independent review found two gaps in the first fix. The TUI/Desktop/dashboard goal path (tui_gateway/prompt_turn.py) still passed every agent's processes to the judge; it now filters by the session's own key like the CLI and gateway loops. And the 30-minute barrier cap (and a pid/timer barrier lifting at all) was only checked lazily, on the next turn, so a parked session with nothing else arriving stayed parked. The CLI idle hook now checks the barrier every 5 s while idle and queues the goal's continuation the moment it lifts. Tests (2 new): an elapsed timed barrier queues the continuation from the idle hook and clears the barrier; an unparked or inactive goal is a no-op. --- cli.py | 1 + hermes_cli/cli_loops_mixin.py | 29 ++++++++++++ tests/cli/test_cli_goal_parked_resume.py | 58 ++++++++++++++++++++++++ tui_gateway/prompt_turn.py | 4 +- 4 files changed, 91 insertions(+), 1 deletion(-) create mode 100644 tests/cli/test_cli_goal_parked_resume.py diff --git a/cli.py b/cli.py index ec6d2b51bc..63a6438ec3 100644 --- a/cli.py +++ b/cli.py @@ -3468,6 +3468,7 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin, CLITuiMix self._check_termios_drift, lambda: self._drain_process_notifications("cli-idle"), self._maybe_fire_loop_tick, + self._maybe_resume_parked_goal, ): with suppress(Exception): step() diff --git a/hermes_cli/cli_loops_mixin.py b/hermes_cli/cli_loops_mixin.py index 4622a60561..1b7c3ce1b7 100644 --- a/hermes_cli/cli_loops_mixin.py +++ b/hermes_cli/cli_loops_mixin.py @@ -386,6 +386,35 @@ class CLILoopsMixin: self._heartbeat_watchdog_started = False threading.Thread(target=_loop, daemon=True, name="heartbeat-watchdog").start() + def _maybe_resume_parked_goal(self) -> None: + """Idle hook run from process_loop: when a parked /goal's barrier has lifted (the process + exited, the timer elapsed, or the wait aged past its cap), queue the continuation so the + loop resumes WITHOUT waiting for an unrelated turn to re-evaluate it. The barrier used to be + checked only lazily, on the next turn; a session with nothing else arriving stayed parked + indefinitely (one run: 3 h 22 min on a grandchild's poller).""" + now = time.time() + if now - getattr(self, "_last_goal_barrier_check", 0.0) < 5.0: + return + self._last_goal_barrier_check = now + try: + if not self._pending_input.empty(): + return + mgr = self._get_goal_manager() + state = getattr(mgr, "state", None) if mgr is not None else None + if state is None or state.status != "active": + return + if not (state.waiting_on_pid is not None or state.waiting_on_session is not None or state.waiting_until): + return # not parked + if mgr.is_waiting(): + return # barrier still holds (is_waiting() also applies the age cap) + prompt = mgr.next_continuation_prompt() + if prompt: + from cli import _DIM, _RST, _cprint + _cprint(f" {_DIM}▶ Goal barrier lifted — resuming.{_RST}") + self._pending_input.put(prompt) + except Exception as exc: + logging.debug("parked-goal resume check failed: %s", exc) + def _maybe_fire_loop_tick(self) -> None: """Idle hook run from process_loop: fire a due /loop wakeup. diff --git a/tests/cli/test_cli_goal_parked_resume.py b/tests/cli/test_cli_goal_parked_resume.py new file mode 100644 index 0000000000..a5a729f342 --- /dev/null +++ b/tests/cli/test_cli_goal_parked_resume.py @@ -0,0 +1,58 @@ +"""A parked /goal resumes from the idle hook once its barrier lifts, without waiting for another turn.""" +import queue +import time +from unittest.mock import patch + +import pytest + +from hermes_cli import goals +from hermes_cli.cli_loops_mixin import CLILoopsMixin + + +@pytest.fixture +def hermes_home(tmp_path, monkeypatch): + from pathlib import Path + home = tmp_path / ".hermes"; home.mkdir() + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(home)) + goals._DB_CACHE.clear() + yield home + goals._DB_CACHE.clear() + + +class _Cli(CLILoopsMixin): + def __init__(self, mgr): + self._pending_input = queue.Queue() + self._mgr = mgr + + def _get_goal_manager(self): + return self._mgr + + +def test_idle_hook_queues_the_continuation_when_a_timed_barrier_has_elapsed(hermes_home): + mgr = goals.GoalManager(session_id="resume-idle") + mgr.set("finish the thing") + mgr.wait_for_seconds(1, reason="cooldown") + cli = _Cli(mgr) + with patch("cli._cprint"), patch("cli._DIM", ""), patch("cli._RST", ""): + cli._maybe_resume_parked_goal() + assert cli._pending_input.empty() # still parked + mgr.state.waiting_until = time.time() - 1 + mgr._save() + cli._last_goal_barrier_check = 0.0 + cli._maybe_resume_parked_goal() + assert not cli._pending_input.empty() # continuation queued + assert "finish the thing" in cli._pending_input.get() + assert mgr.state.waiting_until == 0.0 # barrier cleared + + +def test_idle_hook_is_a_no_op_for_an_unparked_or_inactive_goal(hermes_home): + mgr = goals.GoalManager(session_id="resume-noop") + mgr.set("g") + cli = _Cli(mgr) + cli._maybe_resume_parked_goal() + assert cli._pending_input.empty() + mgr.clear() + cli._last_goal_barrier_check = 0.0 + cli._maybe_resume_parked_goal() + assert cli._pending_input.empty() diff --git a/tui_gateway/prompt_turn.py b/tui_gateway/prompt_turn.py index d098bfc7ed..d9f9b4f07b 100644 --- a/tui_gateway/prompt_turn.py +++ b/tui_gateway/prompt_turn.py @@ -294,7 +294,9 @@ def _goal_followup_after_turn( if session.get("session_key") and (goal_mgr := _active_goal_manager(session)) is not None: try: from hermes_cli.goals import gather_background_processes as _gather_bg - _bg_procs = _gather_bg() + # Only THIS session's processes (TUI turns register under session_key): subagents' + # pollers must not park the parent's goal. Same rule as the CLI and gateway loops. + _bg_procs = _gather_bg(owner_task_id=session.get("session_key") or None) except Exception: _bg_procs = None decision = goal_mgr.evaluate_after_turn( From 50a13d53eb8792829adf20bb7457607a66c6dc69 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 01:42:39 -0700 Subject: [PATCH 026/276] fix(goal): the judge can WAIT on delegated subagents (4 of 5 nudges in one run re-poked an agent that was waiting on workers) JUDGE_SYSTEM_PROMPT allowed WAIT only for "a background process listed below" or a stated backoff. A fan-out orchestrator's in-flight work is delegated subagents, which are not registry processes, so when the agent said "waiting on 4 worker batches, nothing to dispatch" the judge answered CONTINUE ("no background process is listed to gate on"). In the 1,393-agent run 4 of 5 nudges fired 6-153 s after exactly such a turn and each bought a status recap: 19 API calls at ~450K context, ~$4.9, and no work a batch-complete notification would not have produced. The judge now receives an "Active delegations: N batch(es) still running" line (count of live async_delegation records whose parent_session_id is this session, via the existing _session_records selector) and a WAIT branch for it: wait_for_seconds 600-1800 when the response says it is waiting on them with nothing dispatchable. Both goal-loop callers (CLI, gateway) pass the count. Live A/B with the real auxiliary judge on the run's response shape ("4 worker batches still running ... nothing else is dispatchable until these return"): main 3/3 CONTINUE, branch 3/3 WAIT 1200 s. Tests (2): the prompt carries the delegations line only when N > 0 and a wait verdict parses through; count_active_delegations is scoped to the spawning session and ignores completed records. --- gateway/run_goals.py | 6 ++- hermes_cli/cli_loops_mixin.py | 6 ++- hermes_cli/goals.py | 29 ++++++++++- .../hermes_cli/test_goal_judge_delegations.py | 48 +++++++++++++++++++ 4 files changed, 83 insertions(+), 6 deletions(-) create mode 100644 tests/hermes_cli/test_goal_judge_delegations.py diff --git a/gateway/run_goals.py b/gateway/run_goals.py index 011fc12c11..2fe88d27d9 100644 --- a/gateway/run_goals.py +++ b/gateway/run_goals.py @@ -241,12 +241,13 @@ class GatewayGoalsMixin: if mgr is None or not mgr.is_active(): return - _bg_procs = None + _bg_procs, _active_deleg = None, 0 with suppress(Exception): - from hermes_cli.goals import gather_background_processes as _gather_bg + from hermes_cli.goals import count_active_delegations, gather_background_processes as _gather_bg # Only THIS session's processes (gateway turns register under turn_ctx.session_id): # subagents' pollers must not park the parent's goal. _bg_procs = _gather_bg(owner_task_id=getattr(session_entry, "session_id", None) or None) + _active_deleg = count_active_delegations(getattr(session_entry, "session_id", None)) # judge_goal() is a synchronous aux-LLM HTTP call (10-40 s; would block Discord heartbeats). # _run_in_executor_with_context carries the profile secret scope / aux runtime contextvars @@ -254,6 +255,7 @@ class GatewayGoalsMixin: decision = await self._run_in_executor_with_context( lambda: mgr.evaluate_after_turn( final_response or "", user_initiated=True, background_processes=_bg_procs, + active_delegations=_active_deleg, ), ) msg = decision.get("message") or "" diff --git a/hermes_cli/cli_loops_mixin.py b/hermes_cli/cli_loops_mixin.py index 1b7c3ce1b7..1af5ef9720 100644 --- a/hermes_cli/cli_loops_mixin.py +++ b/hermes_cli/cli_loops_mixin.py @@ -560,14 +560,16 @@ class CLILoopsMixin: last_response = self._last_assistant_response_text() if not last_response.strip(): return + _active_deleg = 0 try: - from hermes_cli.goals import gather_background_processes as _gather_bg + from hermes_cli.goals import count_active_delegations, gather_background_processes as _gather_bg # Only THIS session's processes: subagents' pollers must not park the parent's goal. _bg_procs = _gather_bg(owner_task_id=getattr(self, "session_id", None) or None) + _active_deleg = count_active_delegations(getattr(self.agent, "session_id", None)) except Exception: _bg_procs = None decision = mgr.evaluate_after_turn( - last_response, user_initiated=True, background_processes=_bg_procs) + last_response, user_initiated=True, background_processes=_bg_procs, active_delegations=_active_deleg) _print_decision_message(decision) if decision.get("should_continue"): prompt = decision.get("continuation_prompt") diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py index d730230712..ec5d2c6e5e 100644 --- a/hermes_cli/goals.py +++ b/hermes_cli/goals.py @@ -141,6 +141,11 @@ JUDGE_SYSTEM_PROMPT = ( "``wait_on_pid`` (releases on exit only).\n" "- The agent says it is rate-limited / backing off / must wait a fixed " "period — return seconds in ``wait_for_seconds``.\n" + "- The agent has delegated subagents still running (stated below as " + "active delegations) and the response says it is waiting on them with " + "nothing else dispatchable — return ``wait_for_seconds`` between 600 and " + "1800. Their results wake the agent on their own; re-poking it now only " + "produces a status recap.\n" "Picking WAIT parks the loop without burning a turn; it resumes " "automatically when the pid exits or the time elapses. Do NOT pick WAIT " "just because work remains — only when re-poking now would be pure " @@ -159,6 +164,12 @@ JUDGE_SYSTEM_PROMPT = ( "accepted (true=done, false=continue)." ) +# Judge prompt line for live delegated subagents (WAIT-for-seconds vs CONTINUE). +JUDGE_DELEGATIONS_BLOCK_TEMPLATE = ( + "Active delegations: the agent has {count} delegated subagent batch(es) still running; " + "their results are delivered to it automatically when they finish.\n\n" +) + # Judge prompt block listing running background processes (WAIT vs CONTINUE, which pid). JUDGE_BACKGROUND_BLOCK_TEMPLATE = ( "Background processes the agent currently has running (it may be waiting " @@ -860,6 +871,7 @@ def judge_goal( subgoals: Optional[List[str]] = None, background_processes: Optional[List[Dict[str, Any]]] = None, contract: Optional[GoalContract] = None, + active_delegations: int = 0, ) -> Tuple[str, str, bool, Optional[Dict[str, Any]], bool]: """Ask the auxiliary model whether the goal is satisfied. @@ -886,7 +898,8 @@ def judge_goal( common = dict( goal=_truncate(goal, 2000), response=_truncate(last_response, _JUDGE_RESPONSE_SNIPPET_CHARS), - background_block=_render_background_block(background_processes), + background_block=_render_background_block(background_processes) + + (JUDGE_DELEGATIONS_BLOCK_TEMPLATE.format(count=active_delegations) if active_delegations > 0 else ""), current_time=datetime.now(tz=timezone.utc).astimezone().strftime("%Y-%m-%d %H:%M:%S %Z"), ) if contract is not None and not contract.is_empty(): @@ -912,6 +925,17 @@ def judge_goal( return verdict, reason, parse_failed, wait_directive, False +def count_active_delegations(session_id: Optional[str]) -> int: + """Live async delegation batches spawned by this session (fail-safe 0).""" + if not session_id: + return 0 + try: + from tools.async_delegation import _LIVE_STATES, _session_records + return len(_session_records(_LIVE_STATES, "", "", str(session_id))) + except Exception: + return 0 + + def gather_background_processes(task_id: Optional[str] = None, *, owner_task_id: Optional[str] = None) -> List[Dict[str, Any]]: """Fail-safe snapshot of RUNNING ``process_registry`` sessions for the judge; ``[]`` on any error so the loop degrades to its pre-wait-barrier behavior. @@ -1351,6 +1375,7 @@ class GoalManager: def evaluate_after_turn( self, last_response: str, *, user_initiated: bool = True, background_processes: Optional[List[Dict[str, Any]]] = None, + active_delegations: int = 0, ) -> Dict[str, Any]: """Run gates + judge and update state. Return a decision dict (``status``, ``should_continue``, ``continuation_prompt``, ``verdict``, ``reason``, ``message``). Both real user prompts and our @@ -1376,7 +1401,7 @@ class GoalManager: verdict, reason, parse_failed, wait_directive, transport_failed = judge_goal( state.goal, last_response, subgoals=state.subgoals or None, background_processes=background_processes, - contract=state.contract if state.has_contract() else None, + contract=state.contract if state.has_contract() else None, active_delegations=active_delegations, ) state.last_verdict = verdict state.last_reason = reason diff --git a/tests/hermes_cli/test_goal_judge_delegations.py b/tests/hermes_cli/test_goal_judge_delegations.py new file mode 100644 index 0000000000..96275141b2 --- /dev/null +++ b/tests/hermes_cli/test_goal_judge_delegations.py @@ -0,0 +1,48 @@ +"""The goal judge knows about delegated subagents. + +In a fan-out run 4 of 5 /goal nudges fired 6-153 s after a turn that had said "waiting on workers, +nothing to dispatch": the judge prompt had no WAIT branch for delegated subagents (only for registry +processes), so it returned CONTINUE and each nudge bought a status recap (19 API calls, ~$4.9). +""" +from types import SimpleNamespace +from unittest.mock import patch + +from hermes_cli import goals + + +def _resp(content): + return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content=content))]) + + +def test_judge_prompt_states_active_delegations_and_a_wait_branch_for_them(): + seen = {} + + def fake_call_llm(*a, **kw): + seen["prompt"] = kw.get("messages") or a + return _resp('{"verdict": "wait", "wait_for_seconds": 900, "reason": "waiting on workers"}') + + with patch("agent.auxiliary_client.call_llm", side_effect=fake_call_llm): + verdict, _reason, parse_failed, directive, _transport = goals.judge_goal( + "refactor everything", "Waiting on 4 workers; nothing to dispatch.", active_delegations=4) + text = str(seen["prompt"]) + assert "Active delegations: the agent has 4 delegated subagent batch(es) still running" in text + assert "delegated subagents still running" in goals.JUDGE_SYSTEM_PROMPT + assert (verdict, parse_failed) == ("wait", False) and directive.get("seconds") == 900 + + with patch("agent.auxiliary_client.call_llm", side_effect=fake_call_llm): + goals.judge_goal("g", "r", active_delegations=0) + assert "Active delegations" not in str(seen["prompt"]) + + +def test_count_active_delegations_is_scoped_to_the_spawning_session(): + from tools import async_delegation as ad + + fake = { + "a": {"status": "running", "parent_session_id": "root", "session_key": "", "origin_ui_session_id": ""}, + "b": {"status": "completed", "parent_session_id": "root", "session_key": "", "origin_ui_session_id": ""}, + "c": {"status": "running", "parent_session_id": "other", "session_key": "", "origin_ui_session_id": ""}, + } + with patch.object(ad, "_records", fake): + assert goals.count_active_delegations("root") == 1 + assert goals.count_active_delegations("other") == 1 + assert goals.count_active_delegations(None) == 0 From 8477292f0f4e3ee5e9095fa6faf64482790d5621 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:27:45 -0700 Subject: [PATCH 027/276] fix(goal): a delegation WAIT lifts as soon as a batch returns; the TUI passes the delegation count too Independent review found the delegation WAIT was a plain timer: the batch results arrived with 1,199 s left on a 1,200 s wait and nothing re-judged, so integration sat unfinished until the timer ran out. The wait now records how many batches it was set for (GoalState.waiting_on_delegations) and is_waiting() lifts it the moment fewer are live, so the next idle tick resumes with the result in hand. A plain timed wait (no delegations) is unchanged. The TUI/Desktop goal path now passes the live delegation count to the judge like the CLI and gateway do. Live with the real auxiliary judge: WAIT 900 s with 4 live batches; one returns -> is_waiting() False immediately. Test: parked with 4 live, lifted at 3, plain timed wait unaffected. --- hermes_cli/goals.py | 27 ++++++++++++++----- .../hermes_cli/test_goal_judge_delegations.py | 22 +++++++++++++++ tui_gateway/prompt_turn.py | 6 +++-- 3 files changed, 46 insertions(+), 9 deletions(-) diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py index ec5d2c6e5e..0f1d870209 100644 --- a/hermes_cli/goals.py +++ b/hermes_cli/goals.py @@ -439,6 +439,9 @@ class GoalState: waiting_on_pid: Optional[int] = None waiting_on_session: Optional[str] = None waiting_until: float = 0.0 + # Live delegation batches when a timed WAIT was set because of them; the barrier lifts as soon + # as that count drops (a batch returned), not only when the timer runs out. + waiting_on_delegations: int = 0 waiting_reason: Optional[str] = None waiting_since: float = 0.0 contract: GoalContract = field(default_factory=GoalContract) @@ -452,7 +455,7 @@ class GoalState: def from_json(cls, raw: str) -> "GoalState": data = json.loads(raw) raw_subgoals = data.get("subgoals") or [] - ints = {k: int(data.get(k) or 0) for k in ("turns_used", "consecutive_parse_failures", "consecutive_transport_failures")} + ints = {k: int(data.get(k) or 0) for k in ("turns_used", "consecutive_parse_failures", "consecutive_transport_failures", "waiting_on_delegations")} floats = {k: float(data.get(k) or 0.0) for k in ("created_at", "last_turn_at", "waiting_until", "waiting_since")} return cls( goal=data.get("goal", ""), @@ -484,6 +487,7 @@ class GoalState: self.waiting_on_pid = None self.waiting_on_session = None self.waiting_until = 0.0 + self.waiting_on_delegations = 0 self.waiting_reason = None self.waiting_since = 0.0 @@ -1299,13 +1303,17 @@ class GoalManager: raise ValueError("session_id must be a non-empty string") return self._park(reason, waiting_on_session=session_id) - def wait_for_seconds(self, seconds: int, reason: str = "") -> GoalState: - """Park until ``seconds`` from now (backoff/cooldown waits with no process to track).""" + def wait_for_seconds(self, seconds: int, reason: str = "", *, on_delegations: int = 0) -> GoalState: + """Park until ``seconds`` from now (backoff/cooldown waits with no process to track). With + ``on_delegations`` the wait is FOR those live delegation batches: it also lifts as soon as + fewer are live (a batch result came back), so the loop re-judges with the result in hand + instead of sleeping out a 20-minute timer (independent review: results arrived with 1,199 s + left on the timer and nothing re-judged).""" self._require_active() seconds = int(seconds) if seconds <= 0: raise ValueError("seconds must be a positive integer") - return self._park(reason, waiting_until=time.time() + seconds) + return self._park(reason, waiting_until=time.time() + seconds, waiting_on_delegations=max(0, int(on_delegations))) def stop_waiting(self) -> bool: """Clear any active wait barrier (pid / session / time). Returns True if one was cleared.""" @@ -1331,6 +1339,11 @@ class GoalManager: still = _pid_alive(s.waiting_on_pid) elif s.waiting_until: still = time.time() < s.waiting_until + if still and s.waiting_on_delegations > 0: + # Set because of live delegations: lift the moment one of them returned. + live = count_active_delegations(self.session_id) + if live < s.waiting_on_delegations: + still = False else: return False if still and s.waiting_since and s.waiting_until == 0.0 and time.time() - s.waiting_since > _MAX_BARRIER_WAIT_S: @@ -1353,7 +1366,7 @@ class GoalManager: reason = state.waiting_reason or tgt return _decision("active", False, None, "waiting", reason, f"⏳ Goal parked — waiting on {tgt}: {reason}") - def _apply_wait_directive(self, wait_directive: Dict[str, Any], reason: str) -> Dict[str, Any]: + def _apply_wait_directive(self, wait_directive: Dict[str, Any], reason: str, *, active_delegations: int = 0) -> Dict[str, Any]: """Judge said WAIT: set the barrier and park. The counted turn stands (the judge ran) but no continuation fires; the loop resumes once the barrier clears.""" if wait_directive.get("session_id"): @@ -1361,7 +1374,7 @@ class GoalManager: elif wait_directive.get("pid"): tgt = f"pid {self.wait_on(int(wait_directive['pid']), reason=reason).waiting_on_pid}" else: - self.wait_for_seconds(int(wait_directive["seconds"]), reason=reason) + self.wait_for_seconds(int(wait_directive["seconds"]), reason=reason, on_delegations=active_delegations) tgt = f"{wait_directive['seconds']}s" return _decision("active", False, None, "wait", reason, f"⏳ Goal parked (judge) — waiting on {tgt}: {reason}") @@ -1412,7 +1425,7 @@ class GoalManager: state.consecutive_transport_failures = state.consecutive_transport_failures + 1 if transport_failed else 0 if verdict == "wait" and wait_directive: - return self._apply_wait_directive(wait_directive, reason) + return self._apply_wait_directive(wait_directive, reason, active_delegations=active_delegations) # BLOCKED is NOT done: pause so the user sees the judge's reason and can re-scope or override, # instead of burning turns on an unachievable goal or waving it through as complete. diff --git a/tests/hermes_cli/test_goal_judge_delegations.py b/tests/hermes_cli/test_goal_judge_delegations.py index 96275141b2..28d38c3082 100644 --- a/tests/hermes_cli/test_goal_judge_delegations.py +++ b/tests/hermes_cli/test_goal_judge_delegations.py @@ -46,3 +46,25 @@ def test_count_active_delegations_is_scoped_to_the_spawning_session(): assert goals.count_active_delegations("root") == 1 assert goals.count_active_delegations("other") == 1 assert goals.count_active_delegations(None) == 0 + + +def test_a_delegation_wait_lifts_when_a_batch_returns_not_only_when_the_timer_runs_out(tmp_path, monkeypatch): + """Independent-review witness: results arrived with 1,199 s left on a 1,200 s WAIT and nothing + re-judged; integration sat unfinished until the timer ran out.""" + from pathlib import Path + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes")); (tmp_path / ".hermes").mkdir() + goals._DB_CACHE.clear() + mgr = goals.GoalManager(session_id="root-wait") + mgr.set("integrate the rounds") + with patch.object(goals, "count_active_delegations", return_value=4): + mgr.wait_for_seconds(1200, reason="4 batches running", on_delegations=4) + assert mgr.is_waiting() is True # all four still live: parked + with patch.object(goals, "count_active_delegations", return_value=3): + assert mgr.is_waiting() is False # one returned: barrier lifted early + assert mgr.state.waiting_until == 0.0 and mgr.state.waiting_on_delegations == 0 + # a plain timed wait (no delegations) is unaffected by the delegation count + mgr.wait_for_seconds(1200, reason="cooldown") + with patch.object(goals, "count_active_delegations", return_value=0): + assert mgr.is_waiting() is True + goals._DB_CACHE.clear() diff --git a/tui_gateway/prompt_turn.py b/tui_gateway/prompt_turn.py index d9f9b4f07b..cf00abf0d0 100644 --- a/tui_gateway/prompt_turn.py +++ b/tui_gateway/prompt_turn.py @@ -292,15 +292,17 @@ def _goal_followup_after_turn( return goal_followup try: if session.get("session_key") and (goal_mgr := _active_goal_manager(session)) is not None: + _active_deleg = 0 try: - from hermes_cli.goals import gather_background_processes as _gather_bg + from hermes_cli.goals import count_active_delegations, gather_background_processes as _gather_bg # Only THIS session's processes (TUI turns register under session_key): subagents' # pollers must not park the parent's goal. Same rule as the CLI and gateway loops. _bg_procs = _gather_bg(owner_task_id=session.get("session_key") or None) + _active_deleg = count_active_delegations(getattr(session.get("agent"), "session_id", None)) except Exception: _bg_procs = None decision = goal_mgr.evaluate_after_turn( - raw, user_initiated=True, background_processes=_bg_procs) + raw, user_initiated=True, background_processes=_bg_procs, active_delegations=_active_deleg) if verdict_msg := decision.get("message") or "": _emit("status.update", sid, {"kind": "goal", "text": verdict_msg}) if decision.get("should_continue") and ( From ec4c1e0c985578ddbd1e883109167459b5b88706 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:29:28 -0700 Subject: [PATCH 028/276] fix(delegation): validate compression_threshold_tokens; state that it caps the trigger, not the payload Independent review: a YAML `true` coerced to int 1 and gave every child a one-token compression trigger; "200k" silently disabled the default cap. Values are now validated: an int >= 16000 is used, 0/false/null disable on purpose, anything else warns and falls back to the 200K default so a typo never costs money. Config comment and docs say it caps the compaction trigger, not a hard request-size limit. Test: true / "200k" / 5 -> default; 0 / false / None -> disabled; valid ints pass. --- hermes_cli/config_defaults.py | 3 ++- .../test_delegate_child_compression_cap.py | 9 +++++++ tools/delegate_tool.py | 27 +++++++++++++++---- .../docs/user-guide/features/delegation.md | 2 +- 4 files changed, 34 insertions(+), 7 deletions(-) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 8ccb38cf30..b8a1f2a5b9 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1221,7 +1221,8 @@ DEFAULT_CONFIG = { # model otherwise compresses at threshold x window (850K at 0.85) and re-sends a 300-800K # prefix on every call: in one 1,393-agent run 1,373 children never compressed and calls # above 200K context carried ~55% of the bill. Independent of the parent's own threshold. - # 0 = no subagent-specific cap. + # This caps the compaction TRIGGER, not the request payload. A token count >= 16000; 0 disables + # the subagent-specific cap. Other values (true, "200k") are config errors: warned, default used. "compression_threshold_tokens": 200000, # When delegate_task narrows child toolsets, keep the parent's enabled MCP toolsets (so # toolsets=["web"] doesn't strip MCP). false = strict intersection. diff --git a/tests/tools/test_delegate_child_compression_cap.py b/tests/tools/test_delegate_child_compression_cap.py index 024c3f0855..9277a99cb7 100644 --- a/tests/tools/test_delegate_child_compression_cap.py +++ b/tests/tools/test_delegate_child_compression_cap.py @@ -38,3 +38,12 @@ def test_zero_disables_and_an_already_resolved_trigger_is_reclamped(): assert child.context_compressor.threshold_tokens == 850_000 _apply_child_compression_cap(child, {"compression_threshold_tokens": 300_000}) assert child.context_compressor.threshold_tokens == 300_000 + + +def test_config_values_are_validated_not_coerced(): + """Independent-review witnesses: YAML `true` coerced to int 1 (a one-token trigger) and "200k" + silently disabled the cap. Both fall back to the default with a warning; 0/false/null disable.""" + from tools.delegate_tool import _child_compression_cap_tokens as cap + assert cap(True) == 200_000 and cap("200k") == 200_000 and cap(5) == 200_000 + assert cap(0) is None and cap(False) is None and cap(None) is None + assert cap(150_000) == 150_000 and cap(300_000.0) == 300_000 diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index c0660647fd..4b68cd0b49 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -103,6 +103,26 @@ def _open_child_session_db(parent_agent) -> Any: return None +_CHILD_CAP_DEFAULT = 200_000 +_CHILD_CAP_MIN = 16_000 # below this a child compresses on every call; treat as a config error + + +def _child_compression_cap_tokens(raw) -> "int | None": + """Validated ``delegation.compression_threshold_tokens``: an int >= 16000, or None for "no cap". + ``0``/``false``/``null`` disable on purpose. A bool ``true`` (YAML) would coerce to 1 and make + every call compress; a string like ``"200k"`` would silently disable the default. Both are + config errors: warn once and fall back to the default so the mistake never costs money.""" + if raw is None or raw is False or raw == 0: + return None + if isinstance(raw, bool) or not isinstance(raw, (int, float)) or int(raw) < _CHILD_CAP_MIN: + logger.warning( + "delegation.compression_threshold_tokens=%r is not a token count >= %d; using the default %d " + "(set 0 to disable the subagent cap).", raw, _CHILD_CAP_MIN, _CHILD_CAP_DEFAULT, + ) + return _CHILD_CAP_DEFAULT + return int(raw) + + def _apply_child_compression_cap(child, delegation_cfg: dict) -> None: """Cap the child's compaction trigger at ``delegation.compression_threshold_tokens`` (lower of it and any global ``compression.threshold_tokens``). The compressor applies the cap on first window @@ -112,11 +132,8 @@ def _apply_child_compression_cap(child, delegation_cfg: dict) -> None: cc = getattr(child, "context_compressor", None) if not isinstance(cc, ContextCompressor): return - try: - cap = int((delegation_cfg or {}).get("compression_threshold_tokens", 200_000) or 0) - except (TypeError, ValueError): - cap = 0 - if cap <= 0: + cap = _child_compression_cap_tokens((delegation_cfg or {}).get("compression_threshold_tokens", 200_000)) + if cap is None: return existing = cc.threshold_tokens_cap cc.threshold_tokens_cap = min(cap, existing) if isinstance(existing, int) and existing > 0 else cap diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index dcc51599ac..1950755281 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -537,7 +537,7 @@ delegation: When `base_url` points at an Anthropic-compatible endpoint — for example a path ending in `/anthropic`, an Azure Foundry Claude route, or a MiniMax `/anthropic` proxy — `api_mode` is auto-detected as `anthropic_messages` so the subagent uses the right wire format without you setting anything. Set `api_mode` explicitly when the auto-detection guess is wrong (rare). -Subagents compress at an absolute cap, `delegation.compression_threshold_tokens` (default `200000`), applied as the lower of it and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window model would otherwise compress at `threshold × window` (850K at 0.85) and re-send a 300–800K prefix on every call; in a 1,393-agent run, 1,373 children never compressed and calls above 200K context carried about 55% of the bill. It is independent of the parent's own threshold; `0` disables the subagent-specific cap. +Subagents compress at an absolute cap, `delegation.compression_threshold_tokens` (default `200000`), applied as the lower of it and the child's ratio threshold. A brief-driven, disposable worker on a 1M-window model would otherwise compress at `threshold × window` (850K at 0.85) and re-send a 300–800K prefix on every call; in a 1,393-agent run, 1,373 children never compressed and calls above 200K context carried about 55% of the bill. It is independent of the parent's own threshold and caps the compaction *trigger*, not the request payload. Accepted values: a token count of at least 16000, or `0` to disable the subagent-specific cap; anything else (a bare `true`, `"200k"`) is a config error that falls back to the default with a warning. `delegation.request_overrides` works on **all three** resolution branches — direct `base_url`, named `provider`, and pure inherit — so it always takes effect. Top-level keys are API kwargs (e.g. `service_tier`); an `extra_body` sub-dict is merged into the request's `extra_body`. Explicit values merge **over** runtime- or parent-derived overrides: explicit top-level keys win, and `extra_body` is deep-merged one level, so a provider's own request personality (e.g. `thinking: {type: disabled}`) survives unless your key redefines it. See [Configuration → Delegation](../configuration.md#delegation) for details. From 8581b57f4671a67c640d377f02101203e8baf626 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:30:33 -0700 Subject: [PATCH 029/276] fix(terminal): promotion keeps the shell-detachment refusal; the note only promises a notification the session can receive Independent review: an over-cap timeout on `cmd &` was promoted, so the tracked shell exited at once while the payload ran untracked (the exact thing the '&' guidance exists to prevent); and the note promised a notification even on finite sessions where async delivery is disabled and the spawn had already cleared notify_on_complete. The detachment guidance now runs before the promotion decision regardless of timeout, and the note reads the spawn's actual notify_on_complete: notification wording when kept, poll-only wording when the session cannot receive one. Tests (2 new): '&' and nohup with an over-cap timeout still refuse; the note matches the delivery capability. --- .../test_terminal_foreground_timeout_cap.py | 23 +++++++++++++++++ tools/terminal_tool.py | 25 ++++++++++++++----- 2 files changed, 42 insertions(+), 6 deletions(-) diff --git a/tests/tools/test_terminal_foreground_timeout_cap.py b/tests/tools/test_terminal_foreground_timeout_cap.py index 7d437c304b..b43c4941b9 100644 --- a/tests/tools/test_terminal_foreground_timeout_cap.py +++ b/tests/tools/test_terminal_foreground_timeout_cap.py @@ -170,3 +170,26 @@ class TestForegroundMaxTimeoutConstant: timeout_desc = TERMINAL_SCHEMA["parameters"]["properties"]["timeout"]["description"] assert str(FOREGROUND_MAX_TIMEOUT) in timeout_desc assert "background process" in timeout_desc + + +class TestPromotionKeepsTheDetachmentGuard: + def test_over_cap_timeout_with_shell_backgrounding_is_still_refused(self): + """Independent-review witness: a promoted `cmd &` started a tracked shell that exited at once + while the payload ran untracked, defeating the guidance the refusal exists for.""" + from tools.terminal_tool import terminal_tool + + with patch("tools.terminal_tool._get_env_config", return_value=_make_env_config()), \ + patch("tools.terminal_tool._start_cleanup_thread"): + result = json.loads(terminal_tool(command="sleep 5 &", timeout=9999)) + result2 = json.loads(terminal_tool(command="nohup make test", timeout=9999)) + assert "'&' backgrounding" in result["error"] + assert "nohup" in result2["error"] + + def test_note_does_not_promise_a_notification_the_session_cannot_receive(self): + from tools.terminal_tool import _with_promoted_note + + kept = json.loads(_with_promoted_note(json.dumps({"session_id": "proc_x", "error": None, "notify_on_complete": True}), 900)) + assert "arrives as a notification" in kept["promoted_from_foreground"] + dropped = json.loads(_with_promoted_note(json.dumps({"session_id": "proc_x", "error": None, "notify_on_complete": False}), 900)) + assert "cannot receive completion notifications" in dropped["promoted_from_foreground"] + assert "poll" in dropped["promoted_from_foreground"] diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index 6815ad542a..d199c255fd 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -954,12 +954,13 @@ def _plan_execution( # re-sent lower/split/background. Promote to a tracked background process instead; the # caller is told in the result. The `&`/nohup/server guidance below stays a refusal: those # need the command itself rewritten, which the tool cannot do safely. + # The detachment guidance applies whether or not the call is promoted: a promoted `cmd &` + # would start a tracked shell that exits at once while its payload runs untracked. + guidance = _foreground_background_guidance(command) + if guidance: + raise _Rejected(_error_json(guidance, status="error")) if timeout and timeout > FOREGROUND_MAX_TIMEOUT: promoted = timeout - else: - guidance = _foreground_background_guidance(command) - if guidance: - raise _Rejected(_error_json(guidance, status="error")) return _ExecPlan( config=config, env_type=env_type, effective_task_id=effective_task_id, @@ -968,15 +969,25 @@ def _plan_execution( ) +_PROMOTED_NOTE_POLL_ONLY = ( + "Requested foreground timeout {requested}s exceeds the {cap}s cap, so this command was started as a " + "tracked background process instead of being refused. Do NOT re-run it. This session cannot receive " + "completion notifications, so poll it with process(action=\"poll\", session_id=...) until it exits." +) + + def _with_promoted_note(result_json: str, requested_timeout: int) -> str: - """Attach the foreground->background promotion note to a spawn result (unchanged on error).""" + """Attach the foreground->background promotion note to a spawn result (unchanged on error). The + note only promises a notification when the spawn actually kept notify_on_complete (finite sessions + such as one-shot runners cannot route one back; the spawn already said so and cleared the flag).""" try: data = json.loads(result_json) except (TypeError, ValueError): return result_json if not isinstance(data, dict) or data.get("error"): return result_json - data["promoted_from_foreground"] = _PROMOTED_NOTE.format(requested=requested_timeout, cap=FOREGROUND_MAX_TIMEOUT) + template = _PROMOTED_NOTE if data.get("notify_on_complete") else _PROMOTED_NOTE_POLL_ONLY + data["promoted_from_foreground"] = template.format(requested=requested_timeout, cap=FOREGROUND_MAX_TIMEOUT) return json.dumps(data, ensure_ascii=False) @@ -1181,6 +1192,8 @@ def terminal_tool( pty_disabled = pty and _command_requires_pipe_stdin(command) if plan.promoted_from_foreground_timeout is not None: + # Promotion implies notify_on_complete; watch_patterns is a background-only flag the + # caller could not have meant for a foreground call, and the two are exclusive anyway. background, notify_on_complete, watch_patterns = True, True, None if background: result = spawn_background_process( From bb504a2e14c34afce893d74f568a51433f190515 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:55:46 +0530 Subject: [PATCH 030/276] fix(terminal): every Windows drive letter is a host cwd, not just C: (#60962) _HOST_CWD_PREFIXES only listed C:\ and C:/, so a D:\ or E:/ working directory on a Windows host reached docker run -w and the container started in a directory that does not exist. The two prefix scans now go through one _is_host_cwd predicate that matches any drive letter with either slash. --- tests/tools/test_modal_sandbox_fixes.py | 26 ++++++++++++++----------- tools/terminal_tool.py | 4 ++-- tools/terminal_tool_config.py | 14 +++++++++---- 3 files changed, 27 insertions(+), 17 deletions(-) diff --git a/tests/tools/test_modal_sandbox_fixes.py b/tests/tools/test_modal_sandbox_fixes.py index 9baed6aa63..ab0be17287 100644 --- a/tests/tools/test_modal_sandbox_fixes.py +++ b/tests/tools/test_modal_sandbox_fixes.py @@ -317,22 +317,26 @@ class TestHostPrefixList: refactor that moves the constant. """ - def test_all_common_host_prefixes_present_in_constant(self): - """The shared prefix constant must list the common host-only roots.""" - for prefix in ("/Users/", "/home/", "C:\\", "C:/"): - assert prefix in _tt_mod._HOST_CWD_PREFIXES, ( - f"Host prefix {prefix!r} missing from _HOST_CWD_PREFIXES. " - "Container backends need this to avoid using host paths." - ) - def test_all_common_host_paths_flagged_unusable(self): - """A host path under each prefix must be rejected as a container cwd.""" - for host_path in ("/Users/me/proj", "/home/me/proj", - "C:\\Users\\me", "C:/Users/me"): + """A host path under each user root must be rejected as a container cwd; in-sandbox + absolute paths pass.""" + for host_path in ("/Users/me/proj", "/home/me/proj", "C:\\Users\\me", "C:/Users/me"): assert _tt_mod._is_unusable_container_cwd(host_path) is True, ( f"Host path {host_path!r} should be rejected as a container " "cwd but was accepted." ) + for sandbox_path in ("/workspace", "/root/proj", "/srv/app"): + assert _tt_mod._is_unusable_container_cwd(sandbox_path) is False + + def test_any_windows_drive_letter_is_a_host_cwd(self): + """The host-shape predicate is platform-independent data: every drive letter, either slash, + is a host path (#60962). On POSIX ``D:\\proj`` is also non-absolute so the container guard + already rejected it; on a Windows host it IS absolute and only this predicate catches it.""" + from tools.terminal_tool_config import _is_host_cwd + for host_path in ("C:\\Users\\me", "D:\\proj", "e:/work", "Z:\\"): + assert _is_host_cwd(host_path) is True, host_path + for not_host in ("/workspace", "/srv/app", "relative/dir", "C", "C:"): + assert _is_host_cwd(not_host) is False, not_host # ========================================================================= diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index c5e8419822..65f5e7b908 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -43,7 +43,7 @@ from tools.terminal_tool_lifecycle import ( _evict_environment_for_task, cleanup_all_environments, ensure_task_env, ) from tools.terminal_tool_config import ( - _HOST_CWD_PREFIXES, _is_container_backend, _is_unusable_container_cwd, _parse_env_var, + _is_container_backend, _is_host_cwd, _is_unusable_container_cwd, _parse_env_var, _plugin_env_flag, _quiet, _safe_getcwd, _tenv, _tenv_bool, ) from tools.terminal_tool_backends import ( @@ -580,7 +580,7 @@ def _resolve_config_cwd(env_type: str, mount_docker_cwd: bool) -> tuple: if env_type == "docker" and mount_docker_cwd: candidate = os.path.abspath(os.path.expanduser(_tenv("TERMINAL_CWD") or _safe_getcwd())) if ( - any(candidate.startswith(p) for p in _HOST_CWD_PREFIXES) + _is_host_cwd(candidate) or (os.path.isabs(candidate) and os.path.isdir(candidate) and not candidate.startswith(("/workspace", "/root"))) ): host_cwd = candidate diff --git a/tools/terminal_tool_config.py b/tools/terminal_tool_config.py index b1643f7f0f..9bb017040d 100644 --- a/tools/terminal_tool_config.py +++ b/tools/terminal_tool_config.py @@ -9,6 +9,7 @@ so ``tools.terminal_tool.`` keeps resolving (and monkeypatching) as before import logging import json import os +import re from contextlib import contextmanager from typing import Any @@ -51,9 +52,14 @@ def _safe_getcwd() -> str: return _tenv("TERMINAL_CWD") or os.path.expanduser("~") -# Host-cwd prefixes that cannot exist inside a container sandbox (POSIX user -# dirs and Windows drive paths as they leak toward a Linux ``-w`` flag). -_HOST_CWD_PREFIXES = ("/Users/", "/home/", "C:\\", "C:/") +# Host-cwd shapes that cannot exist inside a container sandbox: POSIX user dirs and ANY Windows +# drive path (``D:\\...``, ``e:/...``) as they leak toward a Linux ``-w`` flag. +_HOST_CWD_PREFIXES = ("/Users/", "/home/") +_WINDOWS_DRIVE_RE = re.compile(r"^[A-Za-z]:[\\/]") + + +def _is_host_cwd(path: str) -> bool: + return path.startswith(_HOST_CWD_PREFIXES) or bool(_WINDOWS_DRIVE_RE.match(path)) _CONTAINER_BACKENDS = frozenset({"docker", "singularity", "modal", "daytona", "vercel_sandbox"}) _BUILTIN_BACKENDS = _CONTAINER_BACKENDS | {"local", "ssh", "managed_modal"} @@ -94,7 +100,7 @@ def _is_unusable_container_cwd(cwd: str) -> bool: workdir: ``docker run -w`` needs an absolute in-sandbox path, otherwise the container fails to start (exit 125). Windows drive paths aren't ``isabs`` on POSIX, so they're caught by the prefix check.""" - return bool(cwd) and (cwd.startswith(_HOST_CWD_PREFIXES) or not os.path.isabs(cwd)) + return bool(cwd) and (_is_host_cwd(cwd) or not os.path.isabs(cwd)) def _tenv(name: str, default: str = "") -> str: From 337113c1b8e26ed621a663eb5678822810d2523f Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:55:47 +0530 Subject: [PATCH 031/276] fix(agent): a parent directory the process cannot stat is not a crash while locating .git (#8751) _find_git_root walked cwd's parents with Path.exists(); on hosts where a parent is mode 700 for another user that raised PermissionError out of prompt construction. Treat an unreadable ancestor as 'no .git here'. --- agent/prompt_builder.py | 10 +++++++++- tests/agent/test_prompt_builder.py | 15 +++++++++++++++ 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index 7ee3cde17a..69c7a72906 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -96,7 +96,15 @@ def _scan_context_content(content: str, filename: str) -> str: def _find_git_root(start: Path) -> Optional[Path]: """Nearest ancestor (or *start* itself) containing ``.git``, else None.""" current = start.resolve() - return next((p for p in (current, *current.parents) if (p / ".git").exists()), None) + # A parent the process may not stat (locked-down /home on shared hosts) is "no .git here", not a crash. + return next((p for p in (current, *current.parents) if _exists_or_denied(p / ".git")), None) + + +def _exists_or_denied(path: Path) -> bool: + try: + return path.exists() + except OSError: + return False def _find_hermes_md(cwd: Path) -> Optional[Path]: diff --git a/tests/agent/test_prompt_builder.py b/tests/agent/test_prompt_builder.py index 4e2101b0a8..0e23a75257 100644 --- a/tests/agent/test_prompt_builder.py +++ b/tests/agent/test_prompt_builder.py @@ -586,6 +586,21 @@ class TestFindHermesMd: + def test_unreadable_parent_is_treated_as_no_git_root(self, tmp_path, monkeypatch): + """A parent the process cannot stat (#8751) must not raise out of prompt construction.""" + import os as _os + project = tmp_path / "locked" / "proj" + project.mkdir(parents=True) + real_exists = Path.exists + + def _exists(self): + if self.parent == tmp_path / "locked" and self.name == ".git": + raise PermissionError(13, "Permission denied", str(self)) + return real_exists(self) + + monkeypatch.setattr(Path, "exists", _exists) + assert _find_git_root(project) is None + def test_walks_to_git_root(self, tmp_path): (tmp_path / ".git").mkdir() (tmp_path / ".hermes.md").write_text("root rules") From 60ca93cebd7a770acb1c232d0a369438ed63fb84 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:55:47 +0530 Subject: [PATCH 032/276] fix(doctor): probe the Docker daemon with 'docker version', not 'docker info' (#72927) docker info needs the /info endpoint, which socket proxies such as tecnativa block by default, so doctor reported 'docker daemon not running' against a fully working DOCKER_HOST. docker version hits /version and is what the backend itself probes with. --- hermes_cli/doctor_tools.py | 5 ++++- tests/hermes_cli/test_doctor.py | 16 ++++++++++++++++ 2 files changed, 20 insertions(+), 1 deletion(-) diff --git a/hermes_cli/doctor_tools.py b/hermes_cli/doctor_tools.py index 8c66fb9a7b..2cc8788634 100644 --- a/hermes_cli/doctor_tools.py +++ b/hermes_cli/doctor_tools.py @@ -138,7 +138,10 @@ def _check_docker_backend(terminal_env: str, running_in_container: bool, issues: if not _safe_which("docker"): _fail_and_issue("docker not found", "(required for TERMINAL_ENV=docker)", "Install Docker or change TERMINAL_ENV", issues) else: - _require(_run_ok(["docker", "info"], timeout=10), ("docker", "(daemon running)"), ("docker daemon not running", ""), + # `docker version` hits /version, which socket proxies (tecnativa) allow by default; `docker info` + # needs /info and is commonly blocked, giving a false "daemon not running". The backend itself + # probes with `docker version` too (environments/docker.py). + _require(_run_ok(["docker", "version"], timeout=10), ("docker", "(daemon running)"), ("docker daemon not running", ""), "Start Docker daemon", issues) elif _safe_which("docker"): check_ok("docker", "(optional)") diff --git a/tests/hermes_cli/test_doctor.py b/tests/hermes_cli/test_doctor.py index 9b4cee69bc..74394e9877 100644 --- a/tests/hermes_cli/test_doctor.py +++ b/tests/hermes_cli/test_doctor.py @@ -1742,3 +1742,19 @@ def test_run_doctor_warns_when_lightpanda_binary_missing(monkeypatch, tmp_path): monkeypatch.setattr("tools.browser_lightpanda.find_lightpanda_binary", lambda: None) out = helper._run_doctor_and_capture(monkeypatch, tmp_path) assert "Lightpanda selected but binary not found" in out + + +def test_docker_daemon_probe_uses_version_not_info(monkeypatch): + """`docker info` needs the /info endpoint, which socket proxies commonly block, so doctor reported + "daemon not running" against a working DOCKER_HOST (#72927). `docker version` (/version) is what the + backend itself probes with.""" + from hermes_cli import doctor_tools + + calls: list = [] + monkeypatch.setattr(doctor_tools, "_safe_which", lambda name: "/usr/bin/docker") + monkeypatch.setattr(doctor_tools, "_run_ok", lambda cmd, timeout, **kw: calls.append(cmd) or True) + monkeypatch.setattr(doctor_tools, "_require", lambda *a, **k: None) + + doctor_tools._check_docker_backend("docker", False, []) + + assert calls and calls[0][:2] == ["docker", "version"] From 370ad21deac86c109de243114fe007e75cb9516b Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:55:47 +0530 Subject: [PATCH 033/276] fix(gateway): the skipped-MEDIA warning says whether the file is missing or denied (#100074) 'Skipping unsafe MEDIA directive path' fired for every non-existent file too, sending operators down a security-policy rabbit hole for what was a hallucinated or untranslated path. The warning now names the reason: 'not found on this host' vs 'denied by the delivery policy'. --- gateway/platforms/base.py | 12 +++++++++++- tests/gateway/test_platform_base.py | 11 +++++++++-- 2 files changed, 20 insertions(+), 3 deletions(-) diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index f001a9a0b4..d7eb6dabeb 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -1140,10 +1140,20 @@ def _validated_delivery_path(raw_path, session_key: str, label: str) -> Optional from gateway.media_fetch import fetch_remote_media safe_path = fetch_remote_media(raw) if not safe_path: - logger.warning("Skipping unsafe %s: %s", label, _log_safe_path(raw)) + # Say WHY: a path that does not exist on the host is the common case (a model hallucinated or + # a sandbox path failed to translate) and is not a security rejection. + reason = "not found on this host" if not _existing_regular_file(raw) else "denied by the delivery policy" + logger.warning("Skipping %s (%s): %s", label, reason, _log_safe_path(raw)) return safe_path +def _existing_regular_file(raw: str) -> bool: + try: + return Path(os.path.expanduser(raw)).is_file() + except (OSError, RuntimeError, ValueError): + return False + + SUPPORTED_DOCUMENT_TYPES = { ".pdf": "application/pdf", ".md": "text/markdown", ".txt": "text/plain", ".csv": "text/csv", ".log": "text/plain", ".json": "application/json", ".xml": "application/xml", diff --git a/tests/gateway/test_platform_base.py b/tests/gateway/test_platform_base.py index df54476b24..f9d9c8897e 100644 --- a/tests/gateway/test_platform_base.py +++ b/tests/gateway/test_platform_base.py @@ -1250,8 +1250,15 @@ class TestMediaDeliveryDiagnosability: with caplog.at_level("WARNING"): out = BasePlatformAdapter.filter_media_delivery_paths([(str(outside), False)]) assert out == [] - # The dropped path must be in the log so operators can diagnose it. - assert str(outside) in caplog.text + # The dropped path must be in the log so operators can diagnose it, with the REASON: a policy + # rejection reads differently from a path that simply does not exist (#100074). + assert str(outside) in caplog.text and "denied by the delivery policy" in caplog.text + + def test_missing_file_is_logged_as_not_found_not_unsafe(self, tmp_path, caplog): + ghost = tmp_path / "never-written.pdf" + with caplog.at_level("WARNING"): + assert BasePlatformAdapter.filter_media_delivery_paths([(str(ghost), False)]) == [] + assert "not found on this host" in caplog.text and "unsafe" not in caplog.text def test_crafted_null_path_does_not_abort_batch(self, tmp_path, monkeypatch): """One crafted ~\\x00 path must not drop every other attachment.""" From 08a3c49812d708b6d86e791f7707c96ef1f9d42f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:32:30 -0700 Subject: [PATCH 034/276] fix(file_tools): the rewrite hint reads through the task's backend, skips non-regular and oversized files, and compares in linear time Independent review: the hint read the HOST path even for remote/sandbox backends (a different file), blocked forever on a host FIFO while holding the write lock, and used SequenceMatcher, which is quadratic on repeated lines (a 460 KB same-line file took ~22 s). It now reads the old content via the task's file ops read_file_raw (regular files only, correct backend), skips files above 400 K chars, and compares line multisets (Counter intersection, linear). Reviewer's own probe on this head: 460 KB repeated-line file 22.3 s -> 0.0 s (skipped), 230 KB 46 ms, remote backend gets no host-derived hint, FIFO write returns at once. --- tools/file_tools.py | 32 ++++++++++++++++++++++---------- 1 file changed, 22 insertions(+), 10 deletions(-) diff --git a/tools/file_tools.py b/tools/file_tools.py index ac1c06e22c..0db7d67cae 100644 --- a/tools/file_tools.py +++ b/tools/file_tools.py @@ -716,22 +716,34 @@ _REWRITE_HINT_MIN_CHARS = 20_000 _REWRITE_HINT_MIN_UNCHANGED = 0.80 -def _whole_file_rewrite_hint(resolved: str | None, new_content: str) -> str | None: - """Return a hint when ``new_content`` mostly re-sends what is already on disk at ``resolved``.""" - if not resolved or len(new_content) < _REWRITE_HINT_MIN_CHARS: +# Above this the line diff is skipped: SequenceMatcher on pathological repeated-line files is +# quadratic (a 460 KB same-line file took ~22 s under the write lock). +_REWRITE_HINT_MAX_CHARS = 400_000 + + +def _whole_file_rewrite_hint(task_id: str, resolved: str | None, new_content: str) -> str | None: + """Return a hint when ``new_content`` mostly re-sends what is already at ``resolved``. + + Reads the OLD content through the task's own file ops (``read_file_raw``, the sandbox/remote + backend the write targets), never the host path: on a remote backend the host file is a + different file, and a host FIFO at that path would block the write lock forever. Bounded size + and a line multiset comparison (linear) instead of a sequence diff (quadratic on repeated lines).""" + if not resolved or not (_REWRITE_HINT_MIN_CHARS <= len(new_content) <= _REWRITE_HINT_MAX_CHARS): return None try: - old = Path(resolved).read_text(encoding="utf-8", errors="replace") - except OSError: + result = _get_file_ops(task_id).read_file_raw(resolved) + old = getattr(result, "content", None) + if getattr(result, "error", None) or not isinstance(old, str): + return None + except Exception: return None - if len(old) < _REWRITE_HINT_MIN_CHARS: + if not (_REWRITE_HINT_MIN_CHARS <= len(old) <= _REWRITE_HINT_MAX_CHARS): return None old_lines, new_lines = old.splitlines(), new_content.splitlines() if not old_lines: return None - import difflib - matcher = difflib.SequenceMatcher(None, old_lines, new_lines, autojunk=False) - unchanged = sum(size for _, _, size in matcher.get_matching_blocks()) + from collections import Counter + unchanged = sum((Counter(old_lines) & Counter(new_lines)).values()) ratio = unchanged / max(len(old_lines), len(new_lines)) if ratio < _REWRITE_HINT_MIN_UNCHANGED: return None @@ -775,7 +787,7 @@ def write_file_tool(path: str, content: str, task_id: str = "default", # subagents; different paths stay fully parallel. _lock.enter_context(file_state.lock_path(_resolved)) warnings = _edit_warnings([path], path_to_resolved, task_id) - rewrite_hint = _whole_file_rewrite_hint(_resolved, content) + rewrite_hint = _whole_file_rewrite_hint(task_id, _resolved, content) result_dict = _get_file_ops(task_id).write_file(_resolved or path, content).to_dict() if warnings: result_dict["_warning"] = warnings[0] From d3fdcef13232d01a42e63a90873e783d285e6266 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:36:52 -0700 Subject: [PATCH 035/276] fix(goal): the re-paste pointer applies only to a near-whole re-paste, and on every surface Independent review: substring matching lost the selected objective (after a message offering API or UI work, '/goal ship the API' and '/goal ship the UI' kicked identically), and the gateway and TUI /goal paths still re-sent the full goal. The rule now lives once in hermes_cli.goals.goal_kick_prompt: the pointer replaces the goal only when the goal is >= 400 chars, contained in the last user message, and >= 80% of it. The CLI reads the last user message from its history; the gateway and TUI read it from the SessionDB (fail-safe: goal sent verbatim). Tests: a fragment that selects one option is kicked verbatim; a long goal that is a minority of a longer message is kicked verbatim; all three surfaces call the shared function. --- gateway/slash_commands_goals.py | 7 ++-- hermes_cli/cli_commands_mixin.py | 23 +++--------- hermes_cli/goals.py | 49 ++++++++++++++++++++++++++ tests/cli/test_cli_goal_kick_prompt.py | 36 ++++++++++++++++--- tui_gateway/methods_tools.py | 4 ++- 5 files changed, 93 insertions(+), 26 deletions(-) diff --git a/gateway/slash_commands_goals.py b/gateway/slash_commands_goals.py index a36cfff565..607c9c6d3b 100644 --- a/gateway/slash_commands_goals.py +++ b/gateway/slash_commands_goals.py @@ -196,9 +196,12 @@ class GatewayGoalCommandsMixin: except ValueError as exc: return t("gateway.goal.invalid", error=str(exc)) - # Queue the goal text as an immediate first turn; the post-turn hook takes over after. + # Queue the goal text as an immediate first turn (a short pointer when the user just pasted that + # very text: hermes_cli.goals.goal_kick_prompt); the post-turn hook takes over after. + from hermes_cli.goals import goal_kick_prompt, last_user_message_from_db + kick = goal_kick_prompt(state.goal, last_user_message_from_db(getattr(mgr, "session_id", None))) self._enqueue_goal_turn( - event, state.goal, label="kickoff enqueue", kickoff=True, route=self._adapter_and_key_for(event) + event, kick, label="kickoff enqueue", kickoff=True, route=self._adapter_and_key_for(event) ) base = t("gateway.goal.set", budget=state.max_turns, goal=state.goal) diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index 7f58ad33cd..146f6d85fd 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -2223,26 +2223,11 @@ class CLICommandsMixin: else: self._goal_set(mgr, arg) - # `/goal ` kicks the loop by queueing the goal text as the next user turn. When that text - # is what the user JUST said (a pasted handoff note, a plan the agent already has), re-sending - # it makes the agent spend a turn deciding it is a replay (11 API calls, 6 min, in one run) and - # duplicates ~2k tokens of context. A short pointer starts the loop just as well. - _GOAL_ALREADY_SEEN_KICK = "[Goal set] Continue with the goal you were just given; there is no need to re-read it." - def _goal_kick_prompt(self, goal: str) -> str: - """The goal text, or a short pointer when the last user message already carries it.""" - last_user = "" - for msg in reversed(getattr(self, "conversation_history", None) or []): - if msg.get("role") == "user": - content = msg.get("content") - if isinstance(content, list): - content = " ".join(str(b.get("text", "")) for b in content if isinstance(b, dict)) - last_user = str(content or "") - break - goal_norm = " ".join(goal.split()) - if goal_norm and goal_norm in " ".join(last_user.split()): - return self._GOAL_ALREADY_SEEN_KICK - return goal + """The goal text, or a short pointer when the last user message is essentially that text + (shared rule: ``hermes_cli.goals.goal_kick_prompt``).""" + from hermes_cli.goals import goal_kick_prompt, last_user_message_content + return goal_kick_prompt(goal, last_user_message_content(getattr(self, "conversation_history", None))) def _kick_goal(self, prompt: str) -> bool: """Queue ``prompt`` as the next turn so the loop starts without a separate message.""" diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py index 8d0b24f679..27a0fae759 100644 --- a/hermes_cli/goals.py +++ b/hermes_cli/goals.py @@ -909,6 +909,55 @@ def judge_goal( return verdict, reason, parse_failed, wait_directive, False +# `/goal ` kicks the loop by sending the goal as the next user turn. When that text IS what +# the user just said (a pasted handoff note, a plan the agent already has), re-sending it makes the +# agent spend a turn deciding it is a replay (11 API calls, 6 min, in one run) and duplicates ~2k +# tokens of context. The pointer is used only when the goal is substantially the WHOLE last +# message: a short goal that merely appears inside a longer one ("ship the API" after a message +# offering API or UI work) selects one option, and two different goals must not kick identically. +GOAL_ALREADY_SEEN_KICK = "[Goal set] Continue with the goal you were just given; there is no need to re-read it." +_GOAL_REPASTE_MIN_CHARS = 400 +_GOAL_REPASTE_MIN_SHARE = 0.8 + + +def goal_kick_prompt(goal: str, last_user_message: Any) -> str: + """The goal text, or ``GOAL_ALREADY_SEEN_KICK`` when ``last_user_message`` is essentially that text.""" + content = last_user_message + if isinstance(content, list): + content = " ".join(str(b.get("text", "")) for b in content if isinstance(b, dict)) + goal_norm, last_norm = " ".join(str(goal or "").split()), " ".join(str(content or "").split()) + if ( + len(goal_norm) >= _GOAL_REPASTE_MIN_CHARS + and goal_norm in last_norm + and len(goal_norm) >= _GOAL_REPASTE_MIN_SHARE * len(last_norm) + ): + return GOAL_ALREADY_SEEN_KICK + return goal + + +def last_user_message_content(history: Any) -> Any: + """Content of the newest ``role == "user"`` message in an OpenAI-shaped history, else ``""``.""" + for msg in reversed(history or []): + if isinstance(msg, dict) and msg.get("role") == "user": + return msg.get("content") + return "" + + +def last_user_message_from_db(session_id: Optional[str]) -> Any: + """Newest user message of ``session_id`` from the SessionDB (gateway/TUI surfaces have no live + history object at slash-command time); ``""`` on any error.""" + if not session_id: + return "" + try: + db = _get_session_db() + if db is None: + return "" + rows = db.get_messages(str(session_id), limit=20, latest=True) + return last_user_message_content(rows) + except Exception: + return "" + + def gather_background_processes(task_id: Optional[str] = None) -> List[Dict[str, Any]]: """Fail-safe snapshot of RUNNING ``process_registry`` sessions for the judge; ``[]`` on any error so the loop degrades to its pre-wait-barrier behavior.""" diff --git a/tests/cli/test_cli_goal_kick_prompt.py b/tests/cli/test_cli_goal_kick_prompt.py index ae57abc900..2e15ba44ab 100644 --- a/tests/cli/test_cli_goal_kick_prompt.py +++ b/tests/cli/test_cli_goal_kick_prompt.py @@ -6,6 +6,7 @@ a user message; the kickoff re-sent it and the agent spent 11 API calls / 6 min import queue from hermes_cli.cli_commands_mixin import CLICommandsMixin +from hermes_cli.goals import GOAL_ALREADY_SEEN_KICK def _cli(history): @@ -15,16 +16,16 @@ def _cli(history): return cli -HANDOFF = "HANDOFF: resume round 3 integration.\n - merge r3-16 first\n - then run the full suite" +HANDOFF = "HANDOFF: resume round 3 integration.\n" + "\n".join(f" - step {i}: merge r3-{i} and run its targeted suite, then the full suite" for i in range(8)) def test_goal_that_the_user_just_pasted_kicks_with_a_pointer_not_the_text(): - cli = _cli([{"role": "user", "content": "Here is the plan.\n\n" + HANDOFF + "\n\nGo."}, + cli = _cli([{"role": "user", "content": HANDOFF + "\n\nGo."}, {"role": "assistant", "content": "ok"}]) - assert cli._goal_kick_prompt(HANDOFF) == CLICommandsMixin._GOAL_ALREADY_SEEN_KICK + assert cli._goal_kick_prompt(HANDOFF) == GOAL_ALREADY_SEEN_KICK # block-style content is handled too cli = _cli([{"role": "user", "content": [{"type": "text", "text": HANDOFF}]}]) - assert cli._goal_kick_prompt(HANDOFF) == CLICommandsMixin._GOAL_ALREADY_SEEN_KICK + assert cli._goal_kick_prompt(HANDOFF) == GOAL_ALREADY_SEEN_KICK def test_a_new_goal_or_a_goal_from_an_older_turn_is_kicked_verbatim(): @@ -35,3 +36,30 @@ def test_a_new_goal_or_a_goal_from_an_older_turn_is_kicked_verbatim(): cli = _cli([{"role": "user", "content": HANDOFF}, {"role": "assistant", "content": "done"}, {"role": "user", "content": "now something else"}]) assert cli._goal_kick_prompt(HANDOFF) == HANDOFF + + +def test_a_short_goal_that_selects_one_option_from_the_last_message_is_kicked_verbatim(): + """Independent-review witness: after a message offering API or UI work, `/goal ship the API` and + `/goal ship the UI` produced identical kickoffs. A goal that is a fragment of the last message + carries the selection; only a near-whole re-paste is replaced by the pointer.""" + offer = "I can either ship the API or ship the UI next; which do you want? " * 8 + cli = _cli([{"role": "user", "content": offer}]) + assert cli._goal_kick_prompt("ship the API") == "ship the API" + assert cli._goal_kick_prompt("ship the UI") == "ship the UI" + # a long goal that is only a minority of a much longer message is also kept verbatim + long_goal = "x" * 500 + cli = _cli([{"role": "user", "content": long_goal + " " + "y" * 2000}]) + assert cli._goal_kick_prompt(long_goal) == long_goal + + +def test_gateway_and_tui_surfaces_use_the_same_rule(tmp_path, monkeypatch): + """Independent review: other surfaces still duplicated the full goal. One shared function now.""" + from hermes_cli import goals + long_goal = "HANDOFF " + "step; " * 120 + assert goals.goal_kick_prompt(long_goal, long_goal) == goals.GOAL_ALREADY_SEEN_KICK + assert goals.goal_kick_prompt("ship the API", "ship the API or ship the UI? " * 8) == "ship the API" + # DB-backed lookup fails safe to "" (goal kicked verbatim) when no session/db + assert goals.last_user_message_from_db(None) == "" + import gateway.slash_commands_goals as g, tui_gateway.methods_tools as m # noqa: E401 + import inspect + assert "goal_kick_prompt" in inspect.getsource(g) and "goal_kick_prompt" in inspect.getsource(m) diff --git a/tui_gateway/methods_tools.py b/tui_gateway/methods_tools.py index 9d9e457f75..839c200b44 100644 --- a/tui_gateway/methods_tools.py +++ b/tui_gateway/methods_tools.py @@ -709,7 +709,9 @@ def _cmd_goal(rid, params, session, name, arg): f"⊙ Goal set ({state.max_turns}-turn budget): {state.goal}\n" "I'll keep working until the goal is done, you pause/clear it, or the budget is exhausted.\n" "Controls: /goal status · /goal pause · /goal resume · /goal clear") - return _ok(rid, {"type": "send", "notice": notice, "message": state.goal}) + from hermes_cli.goals import goal_kick_prompt, last_user_message_from_db + kick = goal_kick_prompt(state.goal, last_user_message_from_db(getattr(mgr, "session_id", None))) + return _ok(rid, {"type": "send", "notice": notice, "message": kick}) def _cmd_loop(rid, params, session, name, arg): From 935e10c20ec7816dd688d64b439744b9165a1050 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:02:12 +0530 Subject: [PATCH 036/276] chore: map contributor email for salvaged #44753 (dredozubov) --- contributors/emails/denis.redozubov@gmail.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/denis.redozubov@gmail.com diff --git a/contributors/emails/denis.redozubov@gmail.com b/contributors/emails/denis.redozubov@gmail.com new file mode 100644 index 0000000000..be12b186ef --- /dev/null +++ b/contributors/emails/denis.redozubov@gmail.com @@ -0,0 +1,2 @@ +dredozubov +# PR #44753 salvage From b1ea9bc47ba472a9e1835b83a946b057db7fefae Mon Sep 17 00:00:00 2001 From: Denis Redozubov Date: Sat, 5 Sep 2026 19:02:12 +0530 Subject: [PATCH 037/276] fix(tools): a sandbox that cannot run commands is 'environment unavailable', not 'File not found' (#44750) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _probe_regular_file returned 'missing' for ANY non-zero exit, so a docker container that was still starting, paused, or otherwise unable to exec made read_file / read_file_raw / read_file_bytes report a false 'File not found' — which the model then trusted for the rest of the session. The probe now echoes a sentinel for a genuinely missing path and a non-zero exit without either sentinel is reported as an environment failure with a retry hint; search_files does the same when its existence probe returns neither marker. Salvaged from #44753 by @dredozubov. --- tools/file_operations.py | 27 +++++++++++++++++++++++---- 1 file changed, 23 insertions(+), 4 deletions(-) diff --git a/tools/file_operations.py b/tools/file_operations.py index 5044145bb2..331fdfc9b4 100644 --- a/tools/file_operations.py +++ b/tools/file_operations.py @@ -441,20 +441,29 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): name-based blocklist can't cover a FIFO (a file TYPE at any path); ``[ -f ]`` is a stat (symlinks followed) so it answers without touching content.""" arg = self._escape_shell_arg(path) + # A missing path ECHOES its sentinel: a non-zero exit with no sentinel means the shell itself did + # not run (container still starting, removed out-of-band, transport down) — not a missing file. + # Reporting that as "File not found" made the model trust a false negative for the whole session. stat_result = self._exec( f"if [ -f {arg} ]; then wc -c < {arg} 2>/dev/null; " f"elif [ -e {arg} ]; then echo {NOT_REGULAR_SENTINEL}; " - f"else exit 1; fi") - if stat_result.exit_code != 0: - return 0, "missing" + f"else echo {MISSING_SENTINEL}; fi") stat_output = _strip_terminal_fence_leaks(stat_result.stdout).strip() + if stat_output == MISSING_SENTINEL: + return 0, "missing" if stat_output == NOT_REGULAR_SENTINEL: return 0, "not_regular" + if stat_result.exit_code != 0: + return 0, "env_unavailable" try: return int(stat_output), "ok" except ValueError: return 0, "bad_size" + def _env_unavailable_error(self, path: str) -> ReadResult: + return ReadResult(error=(f"Terminal environment unavailable: could not stat {path} " + "(the sandbox may still be starting or was removed). Retry shortly.")) + def _detect_binary(self, path: str) -> tuple[bool, Optional[bytes]]: """``(is_binary, sample_bytes)`` — byte-layer detection when the transport allows (base64 sample), else the legacy text heuristic (sample is None).""" @@ -812,6 +821,8 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): return self._read_file_missing(path, offset, limit) if status == "not_regular": return self._not_regular_error(path) + if status == "env_unavailable": + return self._env_unavailable_error(path) if self._is_image(path): # never inlined — redirect to the vision tool return self._image_redirect_result(file_size) is_binary, sample_bytes = self._detect_binary(path) @@ -964,6 +975,8 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): return self._suggest_similar_files(path) if status == "not_regular": return self._not_regular_error(path) + if status == "env_unavailable": + return self._env_unavailable_error(path) if self._is_image(path): return ReadResult(is_image=True, is_binary=True, file_size=file_size) is_binary, sample_bytes = self._detect_binary(path) @@ -985,6 +998,8 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): return ReadResult(error=f"File not found: {path}") if status == "not_regular": return self._not_regular_error(path) + if status == "env_unavailable": + return self._env_unavailable_error(path) if status == "bad_size": return ReadResult(error=f"Could not determine file size: {path}") if max_bytes is not None and file_size > max_bytes: @@ -1357,7 +1372,11 @@ class ShellFileOperations(LintMixin, SearchMixin, FileOperations): error=(f"Invalid file search order {order!r}; expected " "'discovery' or 'modified'.")) path = self._expand_path(path) - if "not_found" in self._path_exists_probe(path): + exists_probe = self._path_exists_probe(path) + if "exists" not in exists_probe and "not_found" not in exists_probe: + return SearchResult(error=(f"Terminal environment unavailable: could not stat {path} " + "(the sandbox may still be starting or was removed). Retry shortly.")) + if "not_found" in exists_probe: # Models often pass several paths in one string: search the parts that exist. multi = self._try_multi_path_search( pattern, path, target, file_glob, limit, offset, output_mode, context, order) From 006b1beb00d9d25230571d14277aca3d70e5e11f Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:02:12 +0530 Subject: [PATCH 038/276] test(tools): dead environment surfaces as unavailable across read_file/read_file_raw/read_file_bytes/search; healthy env still reports missing Trimmed from the contributor's suite to two parametrized contracts. Red on origin/main (6 cases). --- tests/tools/test_file_ops_env_not_ready.py | 41 ++++++++++++++++++++++ 1 file changed, 41 insertions(+) create mode 100644 tests/tools/test_file_ops_env_not_ready.py diff --git a/tests/tools/test_file_ops_env_not_ready.py b/tests/tools/test_file_ops_env_not_ready.py new file mode 100644 index 0000000000..0eb74fc81b --- /dev/null +++ b/tests/tools/test_file_ops_env_not_ready.py @@ -0,0 +1,41 @@ +"""A terminal environment that cannot run commands (container still starting, removed out-of-band, +transport down) must surface as an environment error, never as "File not found": the model trusts +a false negative for the rest of the session (#44750).""" + +import pytest + +from tools.file_operations import ShellFileOperations + + +class _DeadEnv: + cwd = "/workspace" + + def execute(self, command: str, cwd=None, **kwargs) -> dict: + return {"output": "Error: container is not running", "returncode": 125} + + +class _HealthyEmptyEnv: + """Runs commands fine; the filesystem has no files.""" + cwd = "/workspace" + + def execute(self, command: str, cwd=None, **kwargs) -> dict: + if command.startswith("if [ -f"): + return {"output": "__hermes_missing__\n", "returncode": 0} + if command.startswith("test -e"): + return {"output": "not_found\n", "returncode": 0} + if command.startswith("echo "): + return {"output": command[5:].strip() + "\n", "returncode": 0} + return {"output": "", "returncode": 1} + + +@pytest.mark.parametrize("op", ["read_file", "read_file_raw", "read_file_bytes", "search"]) +def test_dead_environment_is_reported_as_unavailable_not_missing(op): + ops = ShellFileOperations(_DeadEnv()) + result = getattr(ops, op)("pattern", "state") if op == "search" else getattr(ops, op)("state/notes.md") + assert result.error and "File not found" not in result.error and "environment unavailable" in result.error.lower() + + +@pytest.mark.parametrize("op", ["read_file", "read_file_raw", "read_file_bytes"]) +def test_healthy_environment_still_reports_missing_files(op): + result = getattr(ShellFileOperations(_HealthyEmptyEnv()), op)("state/notes.md") + assert result.error and "File not found" in result.error From 52840577704522ee146dcae553dc3fc231ba6560 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:45:26 +0530 Subject: [PATCH 039/276] chore: map contributor email for salvaged #101678 (StanleyStetson) --- .../emails/24758295+StanleyStetson@users.noreply.github.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/24758295+StanleyStetson@users.noreply.github.com diff --git a/contributors/emails/24758295+StanleyStetson@users.noreply.github.com b/contributors/emails/24758295+StanleyStetson@users.noreply.github.com new file mode 100644 index 0000000000..a549ac9bcb --- /dev/null +++ b/contributors/emails/24758295+StanleyStetson@users.noreply.github.com @@ -0,0 +1,2 @@ +StanleyStetson +# PR #101678 salvage From fca7fdd35d6867fa75c1ea66ab21aea0ec19cb4d Mon Sep 17 00:00:00 2001 From: StanleyStetson <24758295+StanleyStetson@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:45:26 +0530 Subject: [PATCH 040/276] fix(serve): Desktop-over-SSH isolated backends retire themselves when no client is connected (#101626) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit serve --isolated is detached on purpose (setsid/nohup, PPID 1) so it survives the SSH channel closing, and every teardown path lived on the client. A laptop that sleeps mid-session (dark wake reconnects the tunnel, spawns a backend, sleeps again) therefore left a new backend behind every cycle, each one an extra writer on state.db — the multi-writer source behind the WAL corruption incidents. Only for backends started with --ssh-session-token-file: an ASGI wrapper counts accepted WebSocket sessions on every dashboard route without touching the handlers; a watchdog requests a graceful uvicorn exit (WAL checkpoint, exit 0) once no client has been connected for dashboard.ssh_isolated_idle_grace_s (default 15 min) AND no agent turn is running. An unreadable turn state fails closed (the backend stays up). Loopback normally disables the WS ping, but across a tunnel the local socket stays healthy while the far end sleeps, so these backends keep a slow ping (60s / 10 min) that a GIL-holding turn cannot trip. Design, client-count/turn-probe/fail-closed shape and the tunnel-ping rationale from #101678 by @StanleyStetson; this is the slim redo on the decomposed web server (no exclusive home lock / handover protocol — the newcomer never needs to evict an idle incumbent once idle incumbents exit on their own). --- hermes_cli/web_server.py | 33 +++++-- hermes_cli/web_server_idle_exit.py | 138 +++++++++++++++++++++++++++++ 2 files changed, 166 insertions(+), 5 deletions(-) create mode 100644 hermes_cli/web_server_idle_exit.py diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index 1ffe8c6d28..5ffffc9a69 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -1114,7 +1114,7 @@ def _configure_auth_gate( ) -def _build_uvicorn_server(host: str, port: int): +def _build_uvicorn_server(host: str, port: int, *, ssh_isolated: bool = False): """Build the uvicorn ``Config`` + ``Server`` for this bind (reads ``app.state.auth_required``). uvicorn.Server is driven directly (not uvicorn.run) so startup is split from @@ -1145,8 +1145,22 @@ def _build_uvicorn_server(host: str, port: int): except (TypeError, ValueError): return default + # A Desktop-owned SSH-isolated backend is loopback on the SERVER, but the client sits at the far + # end of a tunnel: the local socket stays healthy while the laptop sleeps, so only a slow WS ping + # notices the half-open tunnel (#101626). Its client count is tracked at the ASGI boundary so + # the idle watchdog can retire the backend once nothing is connected. + served_app = app + ping_interval, ping_timeout = (None, None) if _is_loopback else ( + _ws_ping_setting("ws_ping_interval"), _ws_ping_setting("ws_ping_timeout")) + if ssh_isolated: + from hermes_cli.web_server_idle_exit import ( + TUNNEL_WS_PING_INTERVAL_S, TUNNEL_WS_PING_TIMEOUT_S, IdleClientTracker, wrap_asgi_with_ws_tracking) + app.state.ssh_isolated_clients = IdleClientTracker() + served_app = wrap_asgi_with_ws_tracking(app, app.state.ssh_isolated_clients) + ping_interval, ping_timeout = TUNNEL_WS_PING_INTERVAL_S, TUNNEL_WS_PING_TIMEOUT_S + config = uvicorn.Config( - app, host=host, port=port, log_level="warning", + served_app, host=host, port=port, log_level="warning", # Off by default so _ws_client_is_allowed sees the real peer, not # X-Forwarded-For. Gated mode runs behind a TLS terminator and needs # X-Forwarded-Proto for cookie Secure flags. @@ -1154,8 +1168,8 @@ def _build_uvicorn_server(host: str, port: int): # Loopback-only unless the operator trusts a bounded upstream proxy, so # spoofed X-Forwarded-* from arbitrary callers is never honoured. forwarded_allow_ips=_dashboard_forwarded_allow_ips(_dash_cfg), - ws_ping_interval=None if _is_loopback else _ws_ping_setting("ws_ping_interval"), - ws_ping_timeout=None if _is_loopback else _ws_ping_setting("ws_ping_timeout"), + ws_ping_interval=ping_interval, + ws_ping_timeout=ping_timeout, ws_max_size=_DESKTOP_ATTACHMENT_WS_MAX_BYTES, ) return config, uvicorn.Server(config) @@ -1207,6 +1221,15 @@ def _on_server_started( # No-op for standalone `hermes serve` (no HERMES_PARENT_PID). _start_parent_death_watchdog() + # SSH-isolated backends are detached from any parent on purpose (#91668); their liveness signal + # is "does a client still hold a WebSocket" (#101626). + if getattr(app.state, "ssh_isolated_clients", None) is not None: + from hermes_cli.web_server_idle_exit import DEFAULT_IDLE_GRACE_S, start_idle_watchdog + try: + grace = float((load_config().get("dashboard") or {}).get("ssh_isolated_idle_grace_s", DEFAULT_IDLE_GRACE_S)) + except (TypeError, ValueError): + grace = DEFAULT_IDLE_GRACE_S + start_idle_watchdog(server, app.state.ssh_isolated_clients, grace_s=grace) actual_port = _read_bound_port(server, fallback=port) app.state.bound_port = actual_port @@ -1372,7 +1395,7 @@ def start_server( # GHSA-ppp5-vxwm-4cf7). app.state.bound_host = host - config, server = _build_uvicorn_server(host, port) + config, server = _build_uvicorn_server(host, port, ssh_isolated=bool(ssh_session_token)) # Flush-on-kill guard (#94724): chaining SIGTERM/SIGINT handlers persist # in-memory transcripts to state.db before shutdown. Installed BEFORE diff --git a/hermes_cli/web_server_idle_exit.py b/hermes_cli/web_server_idle_exit.py new file mode 100644 index 0000000000..2d39934248 --- /dev/null +++ b/hermes_cli/web_server_idle_exit.py @@ -0,0 +1,138 @@ +"""Idle-exit for Desktop-owned ``hermes serve --isolated`` backends reached over SSH (#101626). + +That backend is deliberately detached (``setsid``/``nohup``, PPID 1) so it survives the SSH channel +closing, and every teardown path lives on the CLIENT. A laptop that sleeps mid-session (dark wake +reconnects the tunnel, spawns a backend, sleeps again) therefore leaves a backend behind every +cycle — each one an extra writer on ``state.db``. The server needs its own liveness signal. + +Two pieces, both scoped to the SSH-isolated case (a session token was handed over via +``--ssh-session-token-file``): + +* An ASGI wrapper counts accepted WebSocket connections (every dashboard WS route: /api/ws, + /api/pty, /api/console, /api/pub, /api/events, /api/audio/speak-stream) without touching the + handlers. When the count has been zero for the grace window and no agent turn is running, the + watchdog asks uvicorn to exit gracefully (WAL checkpoint, exit 0). An indeterminate turn probe + fails closed: the backend stays up. +* Loopback normally disables uvicorn's WS ping (a dead local client sends FIN/RST). Across an SSH + tunnel the local socket is healthy while the far end is asleep, so pings are the only way to notice + a half-open tunnel; the isolated backend keeps a slow ping with a long timeout so a GIL-holding + turn cannot trip it. + +Design and the client-count/turn-probe/fail-closed shape are from #101678 by @StanleyStetson; this +is the slim redo on the decomposed web server. +""" + +from __future__ import annotations + +import logging +import threading +import time +from typing import Callable, Optional + +_log = logging.getLogger(__name__) + +DEFAULT_IDLE_GRACE_S = 900.0 +# Slow enough that a long GIL-holding turn (minutes) cannot trip it, fast enough that a sleeping +# laptop's half-open tunnel is noticed well inside the idle grace window. +TUNNEL_WS_PING_INTERVAL_S = 60.0 +TUNNEL_WS_PING_TIMEOUT_S = 600.0 + + +class IdleClientTracker: + """Live accepted-WebSocket count plus the moment the last client left.""" + + def __init__(self, now: Callable[[], float] = time.monotonic) -> None: + self._now = now + self._lock = threading.Lock() + self._live = 0 + self._last_client_at = now() + + def on_open(self) -> None: + with self._lock: + self._live += 1 + self._last_client_at = self._now() + + def on_close(self) -> None: + with self._lock: + self._live = max(0, self._live - 1) + self._last_client_at = self._now() + + def live_count(self) -> int: + with self._lock: + return self._live + + def idle_for(self) -> float: + with self._lock: + return 0.0 if self._live else self._now() - self._last_client_at + + +def wrap_asgi_with_ws_tracking(app, tracker: IdleClientTracker): + """Count WebSocket sessions at the ASGI boundary: open on the ``websocket.accept`` send, close + when the scope ends. Handlers stay untouched, so a new WS route is tracked automatically.""" + + async def _app(scope, receive, send): + if scope.get("type") != "websocket": + return await app(scope, receive, send) + accepted = False + + async def _send(message): + nonlocal accepted + if message.get("type") == "websocket.accept" and not accepted: + accepted = True + tracker.on_open() + await send(message) + + try: + await app(scope, receive, _send) + finally: + if accepted: + tracker.on_close() + + return _app + + +_probe_failure_logged = False + + +def turn_in_flight() -> Optional[bool]: + """True/False from the gateway's running-session table; None when it cannot be read. The table + lives on ``tui_gateway.server`` (the voice mixin's helper is bound into that namespace). None + keeps the backend alive forever, so the cause is logged once — a silent never-exits would be + the original bug with a new face.""" + global _probe_failure_logged + try: + import tui_gateway.server as gateway + with gateway._sessions_lock: + return any(s.get("running") for s in gateway._sessions.values()) + except Exception: + if not _probe_failure_logged: + _probe_failure_logged = True + _log.warning("idle-exit turn probe unavailable; this backend will not self-retire", exc_info=True) + return None + + +def should_exit_idle(tracker: IdleClientTracker, grace_s: float, + probe: Callable[[], Optional[bool]] = turn_in_flight) -> bool: + """Exit only when no client has been connected for ``grace_s`` AND no turn is provably running. + A probe that cannot answer keeps the process (fail closed).""" + return tracker.idle_for() >= grace_s and probe() is False # idle_for() is 0 while a client is connected + + +def start_idle_watchdog(server, tracker: IdleClientTracker, *, grace_s: float = DEFAULT_IDLE_GRACE_S, + poll_s: float = 15.0, probe: Callable[[], Optional[bool]] = turn_in_flight) -> threading.Thread: + """Daemon thread that sets ``server.should_exit`` once :func:`should_exit_idle` holds.""" + + poll_s = min(poll_s, max(0.5, grace_s / 4)) + + def _loop() -> None: + while not getattr(server, "should_exit", False): + if should_exit_idle(tracker, grace_s, probe): + _log.warning("SSH-isolated backend idle for %.0fs with no client and no running turn; exiting.", + tracker.idle_for()) + server.should_exit = True + return + time.sleep(poll_s) + + thread = threading.Thread(target=_loop, daemon=True, name="ssh-isolated-idle-watchdog") + thread.start() + return thread From 16759c0ec66db93aa155d6c0abb4b8126981e4b2 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:45:26 +0530 Subject: [PATCH 041/276] test(serve): WS sessions counted at the ASGI boundary for any route; exit only after grace with no client and no running turn; only SSH-isolated backends arm it Red on origin/main (module absent). --- tests/hermes_cli/test_web_server_idle_exit.py | 80 +++++++++++++++++++ 1 file changed, 80 insertions(+) create mode 100644 tests/hermes_cli/test_web_server_idle_exit.py diff --git a/tests/hermes_cli/test_web_server_idle_exit.py b/tests/hermes_cli/test_web_server_idle_exit.py new file mode 100644 index 0000000000..f6d68ff776 --- /dev/null +++ b/tests/hermes_cli/test_web_server_idle_exit.py @@ -0,0 +1,80 @@ +"""A Desktop-owned ``serve --isolated`` backend over SSH must retire itself once no client has been +connected for the grace window and no turn is running (#101626): it is detached from any parent on +purpose, so the client count IS its liveness signal.""" + +from fastapi import FastAPI, WebSocket +from starlette.testclient import TestClient + +import hermes_cli.web_server as ws_mod +from hermes_cli.web_server_idle_exit import ( + IdleClientTracker, should_exit_idle, start_idle_watchdog, wrap_asgi_with_ws_tracking) + + +def test_ws_sessions_are_counted_at_the_asgi_boundary_for_any_route(): + app = FastAPI() + + @app.websocket("/api/anything") + async def anything(ws: WebSocket): + await ws.accept() + await ws.receive_text() + await ws.close() + + @app.websocket("/api/refused") + async def refused(ws: WebSocket): + await ws.close(code=4401) # never accepted: must not count + + tracker = IdleClientTracker(now=lambda: 0.0) + client = TestClient(wrap_asgi_with_ws_tracking(app, tracker)) + with client.websocket_connect("/api/anything") as a: + with client.websocket_connect("/api/anything") as b: + assert tracker.live_count() == 2 + b.send_text("bye") + assert tracker.live_count() == 1 + a.send_text("bye") + assert tracker.live_count() == 0 + try: + with client.websocket_connect("/api/refused"): + pass + except Exception: + pass + assert tracker.live_count() == 0 # a refused (never accepted) upgrade is not a client + + +def test_exit_only_after_grace_with_no_client_and_no_running_turn(): + clock = {"t": 0.0} + tracker = IdleClientTracker(now=lambda: clock["t"]) + grace = 900.0 + assert should_exit_idle(tracker, grace, probe=lambda: False) is False # just started + clock["t"] = 901.0 + assert should_exit_idle(tracker, grace, probe=lambda: False) is True + assert should_exit_idle(tracker, grace, probe=lambda: True) is False # turn running + assert should_exit_idle(tracker, grace, probe=lambda: None) is False # indeterminate: fail closed + tracker.on_open() + clock["t"] = 5000.0 + assert should_exit_idle(tracker, grace, probe=lambda: False) is False # a client is connected + tracker.on_close() + assert should_exit_idle(tracker, grace, probe=lambda: False) is False # grace restarts on close + clock["t"] = 5000.0 + grace + 1 + assert should_exit_idle(tracker, grace, probe=lambda: False) is True + + +def test_watchdog_sets_should_exit_and_only_arms_for_ssh_isolated_backends(monkeypatch): + class _Server: + should_exit = False + + server = _Server() + clock = iter([0.0, 0.0, 10_000.0, 10_000.0, 10_000.0, 10_000.0]) + tracker = IdleClientTracker(now=lambda: next(clock, 10_000.0)) + start_idle_watchdog(server, tracker, grace_s=1.0, poll_s=0.01, probe=lambda: False).join(timeout=5) + assert server.should_exit is True + + # The uvicorn builder wraps the app + arms tunnel pings ONLY when a session token was handed over. + monkeypatch.setattr(ws_mod.app.state, "auth_required", False, raising=False) + plain, _ = ws_mod._build_uvicorn_server("127.0.0.1", 0) + assert plain.ws_ping_interval is None and getattr(ws_mod.app.state, "ssh_isolated_clients", None) is None + try: + isolated, _ = ws_mod._build_uvicorn_server("127.0.0.1", 0, ssh_isolated=True) + assert isolated.ws_ping_interval and isolated.ws_ping_timeout > isolated.ws_ping_interval + assert isinstance(ws_mod.app.state.ssh_isolated_clients, IdleClientTracker) + finally: + ws_mod.app.state._state.pop("ssh_isolated_clients", None) # process-global app: never leak the tracker From f77e2a57ed7dc8dce608a35c1b6e19afc123cf7e Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 19:45:26 +0530 Subject: [PATCH 042/276] docs(dashboard): document ssh_isolated_idle_grace_s --- website/docs/user-guide/configuration.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 393ebf4555..0872b83e15 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -2780,6 +2780,7 @@ dashboard: ws_ping_interval: 20.0 # Non-loopback WebSocket keepalive ping interval (seconds) ws_ping_timeout: 20.0 # Non-loopback WebSocket keepalive pong timeout (seconds) ws_orphan_reap_grace_s: 20.0 # Grace before a WS-detached session is reaped (seconds) + ssh_isolated_idle_grace_s: 900.0 # Desktop-over-SSH backend exits after this long with no client and no running turn ws_orphan_activity_stale_s: 600.0 # Activity idle bound before a detached RUNNING turn is interrupted (seconds) startup_orphan_sweep: true # Close session rows orphaned by a dead gateway process at boot ``` @@ -2790,6 +2791,7 @@ dashboard: - `trusted_proxies` — IP addresses or bounded CIDR networks allowed to supply `X-Forwarded-Proto` and `X-Forwarded-For`. Loopback remains trusted automatically. Configure this when the TLS reverse proxy connects from another container or host. Prefer the proxy's exact IP; use a small dedicated network only when its address is dynamic. Wildcards and `/0` networks are rejected. - `oauth` / `basic_auth` / `drain_auth` — auth provider config read by the bundled dashboard-auth plugins. The drain secret itself is **not** set here; it's provisioned via the `HERMES_DASHBOARD_DRAIN_SECRET` env var. See [Web Dashboard](/user-guide/features/web-dashboard) for full auth setup. - `ws_ping_interval` / `ws_ping_timeout` — WebSocket keepalive tuning for non-loopback binds (loopback connections never ping). Raise these on high-latency links (Tailscale, distant SSH tunnels) where the 20 s defaults can manufacture spurious 1006 disconnects. +- `ssh_isolated_idle_grace_s` (default `900`) — a Desktop-owned `hermes serve --isolated` backend reached over SSH is detached from the SSH session on purpose, so a laptop that sleeps mid-connection cannot tear it down; each dark-wake reconnect used to leave another backend holding `state.db`. The backend now retires itself once no client WebSocket has been connected for this long and no agent turn is running (a turn keeps it alive; an unreadable turn state keeps it alive too). Set high if you rely on a detached backend finishing long work after the laptop sleeps. Such backends also send a slow WebSocket ping (60 s, 10 min timeout) so a half-open tunnel is noticed. - `ws_orphan_reap_grace_s` — how long a WS-detached session waits before the orphan reaper collects it. Raise alongside the keepalive values if clients reconnect slowly. (`HERMES_TUI_WS_ORPHAN_REAP_GRACE_S` remains as an internal override.) - `ws_orphan_activity_stale_s` (default `600`) — how long a detached **running** turn's activity clock (the same clock the `agent.turn_liveness` watchdog samples: API waits, stream tokens, tool heartbeats) must be idle before the orphan reaper interrupts it. A client-absent turn that is still actively producing keeps running to completion detached — closing the laptop, backgrounding the mobile app, or a desktop update no longer cancels healthy long turns; only a genuinely wedged turn is interrupted. Set `0` to interrupt at the grace window regardless of activity (old behavior). - `startup_orphan_sweep` (default `true`) — the WS-orphan reap timer above is in-process, so a gateway restart (update, crash, systemd) before it fires leaves the session row open forever — phantom "active" work in `/resume` and dashboards. On every gateway boot — both the stdio TUI (`entry.main`) and the desktop/dashboard WebSocket sidecar (`handle_ws`) — rows with source `tui` / `desktop` / `subagent` whose start time **and** newest message are both older than the session TTL (`HERMES_TUI_SESSION_TTL_S`, default 6 hours) are closed with `end_reason: startup_orphan_reap`. Messaging-platform sessions (Telegram, Discord, …) are never touched, live in-memory sessions (a client that already resumed) are excluded, and swept sessions remain resumable. From 1bfbe19f5fec5966411e9b7269fb8773f2899805 Mon Sep 17 00:00:00 2001 From: fangliquanflq Date: Sat, 5 Sep 2026 20:43:39 +0530 Subject: [PATCH 043/276] fix(file_sync): Windows sync_back no longer drops remote edits (#76267) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two stacked Windows-only failures in FileSyncManager.sync_back: 1. The staging tar was a NamedTemporaryFile held open while the backend's bulk_download_fn reopened the same path for writing — PermissionError on Windows (exclusive handle). mkstemp + close + unlink in finally instead. 2. Staged member keys came from os.path.relpath and mapping parents from Path(remote).parent — both stringify with backslashes on Windows, so no staged file ever matched its POSIX remote key and every edit was skipped as "no host mapping". Keys are now built with as_posix(); parents with posixpath.dirname; the inferred host path is joined with the host's Path. (cherry picked from commit e8cab109db, re-applied on the current file) --- tools/environments/file_sync.py | 24 +++++++++++++++++------- 1 file changed, 17 insertions(+), 7 deletions(-) diff --git a/tools/environments/file_sync.py b/tools/environments/file_sync.py index c9cb048e58..766edb13b5 100644 --- a/tools/environments/file_sync.py +++ b/tools/environments/file_sync.py @@ -302,12 +302,16 @@ class FileSyncManager: except Exception: file_mapping = [] - with tempfile.NamedTemporaryFile(suffix=".tar") as tf: - self._bulk_download_fn(Path(tf.name)) + # mkstemp + close: NamedTemporaryFile keeps an exclusive handle on Windows, so the + # backend's open(dest, "wb") / write_bytes on the same path raised PermissionError. + fd, tar_path = tempfile.mkstemp(suffix=".tar") + os.close(fd) + try: + self._bulk_download_fn(Path(tar_path)) # A misbehaving sandbox could produce an arbitrarily large tar. try: - tar_size = os.path.getsize(tf.name) + tar_size = os.path.getsize(tar_path) except OSError: tar_size = 0 if tar_size > _SYNC_BACK_MAX_BYTES: @@ -317,7 +321,7 @@ class FileSyncManager: return with tempfile.TemporaryDirectory(prefix="hermes-sync-back-") as staging: - with tarfile.open(tf.name) as tar: + with tarfile.open(tar_path) as tar: tar.extractall(staging, filter="data") upload_only = self._upload_only_host_paths | _credential_host_paths() @@ -325,13 +329,19 @@ class FileSyncManager: for dirpath, _dirnames, filenames in os.walk(staging): for fname in filenames: staged_file = os.path.join(dirpath, fname) - remote_path = "/" + os.path.relpath(staged_file, staging) + # Remote keys are POSIX; relpath uses host separators (backslashes on Windows). + remote_path = "/" + Path(os.path.relpath(staged_file, staging)).as_posix() applied += self._apply_staged_file(staged_file, remote_path, file_mapping, upload_only) if applied: logger.info("sync_back: applied %d changed file(s)", applied) else: logger.debug("sync_back: no remote changes detected") + finally: + try: + os.unlink(tar_path) + except OSError: + pass def _apply_staged_file( self, staged_file: str, remote_path: str, file_mapping: list[tuple[str, str]], upload_only_host_paths: set[str], @@ -377,9 +387,9 @@ class FileSyncManager: for host, remote in file_mapping or []: if self._is_upload_only_host_path(host, upload_only_host_paths): continue - remote_dir = str(Path(remote).parent) + remote_dir = posixpath.dirname(remote) # remote paths are POSIX even on a Windows host if remote_path.startswith(remote_dir + "/"): - return str(Path(host).parent) + remote_path[len(remote_dir):] + return str(Path(host).parent / remote_path[len(remote_dir) + 1:]) return None @staticmethod From 431978d47b8018d6334e02c681bbdb70d1435cbb Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:43:39 +0530 Subject: [PATCH 044/276] test(file_sync): backend can reopen the staging tar for writing; POSIX remote keys match on a Windows host (windows_only) --- tests/tools/test_file_sync_back.py | 41 ++++++++++++++++++++++++++++++ 1 file changed, 41 insertions(+) diff --git a/tests/tools/test_file_sync_back.py b/tests/tools/test_file_sync_back.py index 5b29010795..5075dd3aea 100644 --- a/tests/tools/test_file_sync_back.py +++ b/tests/tools/test_file_sync_back.py @@ -457,3 +457,44 @@ class TestSyncBackSizeCap: # Default cap (2 GiB) is far above our tiny tar; extraction should proceed mgr.sync_back(hermes_home=tmp_path / ".hermes") assert Path(host_file).read_bytes() == b"remote_version" + + +class TestSyncBackWindowsHost: + """#76267: sync_back on a Windows host. The staging tar must be reopenable for writing by + the backend (NamedTemporaryFile held an exclusive handle → PermissionError), and remote + keys/parents must stay POSIX (relpath/Path stringify with backslashes on Windows, so no + staged file ever matched its mapping and every edit was dropped as "no host mapping").""" + + def test_backend_can_reopen_the_tar_path_for_writing(self, tmp_path): + """Contract on every host: the download callback receives a path nothing else holds open.""" + seen = {} + + def download(dest: Path) -> None: + buf = io.BytesIO() + with tarfile.open(fileobj=buf, mode="w") as tar: + info = tarfile.TarInfo(name="root/.hermes/skill.py") + info.size = 2 + tar.addfile(info, io.BytesIO(b"v2")) + with open(dest, "wb") as fh: # the SSH/Modal backends write exactly like this + fh.write(buf.getvalue()) + seen["dest"] = dest + + host_file = tmp_path / "host" / "skill.py" + _write_file(host_file, b"v1") + mgr = _make_manager(tmp_path, [(str(host_file), "/root/.hermes/skill.py")], bulk_download_fn=download) + mgr._pushed_hashes["/root/.hermes/skill.py"] = _sha256_bytes(b"v1") + mgr.sync_back(hermes_home=tmp_path / ".hermes") + assert host_file.read_bytes() == b"v2" + assert not seen["dest"].exists() # staging tar removed after use + + @pytest.mark.windows_only + def test_posix_remote_keys_match_on_windows(self, tmp_path): + host_file = tmp_path / "host" / "skill.py" + _write_file(host_file, b"v1") + mapping = [(str(host_file), "/root/.hermes/skills/a/skill.py")] + mgr = _make_manager(tmp_path, mapping, bulk_download_fn=_make_download_fn({ + "root/.hermes/skills/a/skill.py": b"v2", "root/.hermes/skills/a/new.md": b"new"})) + mgr._pushed_hashes["/root/.hermes/skills/a/skill.py"] = _sha256_bytes(b"v1") + mgr.sync_back(hermes_home=tmp_path / ".hermes") + assert host_file.read_bytes() == b"v2" # relpath key was 'root\\.hermes\\...' → skipped + assert (tmp_path / "host" / "new.md").read_bytes() == b"new" # _infer_host_path parent match From 256bd1adc9612ce81cf545e4b5e1a977f128af71 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:54:30 +0530 Subject: [PATCH 045/276] feat(docker): terminal.docker_snap_compat opt-out for snap-packaged Docker under AppArmor (#9730) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit On hosts where Docker ships as a snap (Ubuntu cloud images / Azure VMs), the snap's AppArmor confinement turns two sandbox hardening flags into a dead container at start: `--init` fails with "exec /sbin/docker-init: operation not permitted" and `--security-opt no-new-privileges` then fails every exec the same way ("exec /usr/bin/sleep: operation not permitted"). This is snapd LP#1908448 — not probeable from the client, and docker_extra_args cannot remove flags we add. `terminal.docker_snap_compat: true` drops exactly those two flags; cap-drop ALL, the tmpfs hardening, PID limits and the privdrop caps are unchanged, and a warning is logged at container start. Bridged everywhere the other docker_* keys are (CLI env map, gateway env map, `hermes config set` sync, terminal_tool env read, the shared container_config shaper, DEFAULT_CONFIG). --- cli.py | 1 + gateway/run.py | 1 + hermes_cli/config.py | 2 +- hermes_cli/config_defaults.py | 4 ++++ tools/environments/docker.py | 25 ++++++++++++++++++------- tools/terminal_tool.py | 1 + tools/terminal_tool_backends.py | 3 ++- 7 files changed, 28 insertions(+), 9 deletions(-) diff --git a/cli.py b/cli.py index ec6d2b51bc..9fe19f1296 100644 --- a/cli.py +++ b/cli.py @@ -293,6 +293,7 @@ _TERMINAL_ENV_MAPPINGS = { "ssh_host", "ssh_user", "ssh_port", "ssh_key", "container_cpu", "container_memory", "container_disk", "container_persistent", "docker_volumes", "docker_env", "docker_extra_args", "docker_shm_size", "docker_mount_cwd_to_workspace", "docker_network", "docker_run_as_host_user", + "docker_snap_compat", "docker_persist_across_processes", "docker_shared_container_key", "docker_orphan_reaper", "sandbox_dir", "persistent_shell", ) diff --git a/gateway/run.py b/gateway/run.py index f59ae79a82..0a62103a77 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -1841,6 +1841,7 @@ def _bridge_terminal_config_to_env(_terminal_cfg: dict) -> None: "docker_mount_cwd_to_workspace": "TERMINAL_DOCKER_MOUNT_CWD_TO_WORKSPACE", "docker_network": "TERMINAL_DOCKER_NETWORK", "docker_run_as_host_user": "TERMINAL_DOCKER_RUN_AS_HOST_USER", + "docker_snap_compat": "TERMINAL_DOCKER_SNAP_COMPAT", "docker_persist_across_processes": "TERMINAL_DOCKER_PERSIST_ACROSS_PROCESSES", "docker_shared_container_key": "TERMINAL_DOCKER_SHARED_CONTAINER_KEY", "docker_orphan_reaper": "TERMINAL_DOCKER_ORPHAN_REAPER", diff --git a/hermes_cli/config.py b/hermes_cli/config.py index e384a4dca4..c6e9d5a717 100644 --- a/hermes_cli/config.py +++ b/hermes_cli/config.py @@ -2055,7 +2055,7 @@ TERMINAL_CONFIG_ENV_MAP = { "daytona_image", "vercel_runtime", "ssh_host", "ssh_user", "ssh_port", "ssh_key", "container_cpu", "container_memory", "container_disk", "container_persistent", "docker_volumes", "docker_env", "docker_mount_cwd_to_workspace", "docker_network", - "docker_extra_args", "docker_shm_size", "docker_run_as_host_user", + "docker_extra_args", "docker_shm_size", "docker_run_as_host_user", "docker_snap_compat", "docker_persist_across_processes", "docker_shared_container_key", "docker_orphan_reaper", "sandbox_dir", "persistent_shell")}} diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 234ef5dcc1..1f98ab6024 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -332,6 +332,10 @@ DEFAULT_CONFIG = { # default for images whose entrypoints must start as root (e.g. the bundled Hermes image, # which drops to `hermes` via s6-setuidgid). When on, SETUID/SETGID caps are omitted. "docker_run_as_host_user": False, + # Snap-packaged Docker under AppArmor (Ubuntu cloud images; LP#1908448) refuses to exec + # anything under `--init` or `--security-opt no-new-privileges` ("operation not + # permitted"). True drops those two flags; every other hardening stays. See #9730. + "docker_snap_compat": False, # Trusted profiles sharing one Docker container identity; empty = per-profile boundary. "docker_shared_container_key": "", # Keep a long-lived bash shell across execute() calls so cwd/env/shell variables survive. diff --git a/tools/environments/docker.py b/tools/environments/docker.py index 99761a52bc..c53ca663cc 100644 --- a/tools/environments/docker.py +++ b/tools/environments/docker.py @@ -239,7 +239,6 @@ _BASE_SECURITY_ARGS = [ "--cap-add", "DAC_OVERRIDE", "--cap-add", "CHOWN", "--cap-add", "FOWNER", - "--security-opt", "no-new-privileges", "--tmpfs", "/tmp:rw,nosuid,size=512m", "--tmpfs", "/var/tmp:rw,noexec,nosuid,size=256m"] @@ -295,9 +294,15 @@ _PRIVDROP_CAP_ARGS = ["--cap-add", "SETUID", "--cap-add", "SETGID"] _S6_INIT_ENTRYPOINTS = ("/init", "/package/admin/s6-overlay/command/init") -def _build_security_args(run_as_host_user: bool, run_exec: bool = False) -> list[str]: - """Security/cap/tmpfs args for the privilege mode; ``run_exec`` mounts /run exec for s6 images.""" - args = list(_BASE_SECURITY_ARGS) + list(_RUN_TMPFS_EXEC if run_exec else _RUN_TMPFS_NOEXEC) +_NO_NEW_PRIVILEGES_ARGS = ["--security-opt", "no-new-privileges"] + + +def _build_security_args(run_as_host_user: bool, run_exec: bool = False, snap_compat: bool = False) -> list[str]: + """Security/cap/tmpfs args for the privilege mode; ``run_exec`` mounts /run exec for s6 images. + ``snap_compat`` drops no-new-privileges: snap-packaged Docker's AppArmor profile turns it into + "exec: operation not permitted" for every process in the container (#9730, LP#1908448).""" + args = list(_BASE_SECURITY_ARGS) + ([] if snap_compat else list(_NO_NEW_PRIVILEGES_ARGS)) + args += list(_RUN_TMPFS_EXEC if run_exec else _RUN_TMPFS_NOEXEC) return args if run_as_host_user else args + list(_PRIVDROP_CAP_ARGS) @@ -501,7 +506,8 @@ class DockerEnvironment(BaseEnvironment): extra_args: list = None, persist_across_processes: bool = True, shm_size: str = _DEFAULT_SHM_SIZE, - shared_container_key: str = ""): + shared_container_key: str = "", + snap_compat: bool = False): if cwd == "~": cwd = "/root" super().__init__(cwd=cwd, timeout=timeout) @@ -547,7 +553,12 @@ class DockerEnvironment(BaseEnvironment): "Docker: image %s uses /init (s6-overlay) as entrypoint — " "skipping --init and mounting /run with exec.", image) - security_args = _build_security_args(run_as_host_user and bool(user_args), run_exec=image_uses_s6_init) + security_args = _build_security_args( + run_as_host_user and bool(user_args), run_exec=image_uses_s6_init, snap_compat=snap_compat) + self._snap_compat = snap_compat + if snap_compat: + logger.warning( + "docker_snap_compat: running without --init and no-new-privileges (snap Docker under AppArmor)") logger.info("Docker volume_args: %s", volume_args) # docker_extra_args go last so they can override defaults. @@ -751,7 +762,7 @@ class DockerEnvironment(BaseEnvironment): # own /init PID 1, so adding --init there creates two competing inits and breaks startup # (#34628). self._docker_exe, "run", "-d", - *([] if self._image_uses_s6_init else ["--init"]), + *([] if self._image_uses_s6_init or self._snap_compat else ["--init"]), "--name", name, *label_args, "-w", workdir, diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index 65f5e7b908..6aed41dc62 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -654,6 +654,7 @@ def _get_env_config() -> Dict[str, Any]: "docker_volumes": docker_volumes, "docker_env": docker_env, "docker_run_as_host_user": _tenv_bool("TERMINAL_DOCKER_RUN_AS_HOST_USER", "false"), + "docker_snap_compat": _tenv_bool("TERMINAL_DOCKER_SNAP_COMPAT", "false"), "docker_network": _tenv_bool("TERMINAL_DOCKER_NETWORK", "true"), "docker_extra_args": docker_extra_args, "docker_shm_size": docker_shm_size, diff --git a/tools/terminal_tool_backends.py b/tools/terminal_tool_backends.py index 8eb38cb67c..63a3dd1925 100644 --- a/tools/terminal_tool_backends.py +++ b/tools/terminal_tool_backends.py @@ -40,7 +40,7 @@ _CONTAINER_KEYS = ( ("docker_volumes", []), ("docker_mount_cwd_to_workspace", False), ("docker_forward_env", []), ("docker_env", {}), ("docker_run_as_host_user", False), ("docker_extra_args", []), ("docker_shm_size", "1g"), ("docker_network", True), ("docker_persist_across_processes", True), - ("docker_shared_container_key", ""), ("docker_orphan_reaper", True), + ("docker_shared_container_key", ""), ("docker_orphan_reaper", True), ("docker_snap_compat", False), ) _DOCKER_KWARGS = ( ("volumes", "docker_volumes", []), ("auto_mount_cwd", "docker_mount_cwd_to_workspace", False), @@ -48,6 +48,7 @@ _DOCKER_KWARGS = ( ("run_as_host_user", "docker_run_as_host_user", False), ("network", "docker_network", True), ("extra_args", "docker_extra_args", []), ("persist_across_processes", "docker_persist_across_processes", True), ("shared_container_key", "docker_shared_container_key", ""), ("shm_size", "docker_shm_size", "1g"), + ("snap_compat", "docker_snap_compat", False), ) From b800fdf9a80bbb0948feb078088d72cb8dd28f80 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:54:30 +0530 Subject: [PATCH 046/276] test(docker): snap_compat drops only --init and no-new-privileges; key bridged on every config path --- tests/tools/test_docker_environment.py | 22 ++++++++++++++++++++ tests/tools/test_terminal_config_env_sync.py | 9 ++++++++ 2 files changed, 31 insertions(+) diff --git a/tests/tools/test_docker_environment.py b/tests/tools/test_docker_environment.py index b92f2dcda1..cafc174e8d 100644 --- a/tests/tools/test_docker_environment.py +++ b/tests/tools/test_docker_environment.py @@ -56,6 +56,7 @@ def _make_dummy_env(**kwargs): persist_across_processes=kwargs.get("persist_across_processes", True), shared_container_key=kwargs.get("shared_container_key", ""), shm_size=kwargs.get("shm_size", docker_env._DEFAULT_SHM_SIZE), + snap_compat=kwargs.get("snap_compat", False), ) @@ -485,6 +486,27 @@ def test_security_args_include_setuid_setgid_for_privdrop(monkeypatch): assert "SETGID" in added, "SETGID cap missing — image privilege-drop will fail" +def test_snap_compat_drops_only_init_and_no_new_privileges(monkeypatch): + """#9730: snap-packaged Docker under AppArmor turns ``--init`` and ``no-new-privileges`` into + "exec: operation not permitted" for every process in the container. The opt-out drops exactly + those two flags; cap-drop, tmpfs hardening and the privdrop caps are unchanged.""" + monkeypatch.setattr(docker_env, "find_docker", lambda: "/usr/bin/docker") + + def run_args(**kw): + calls = _mock_subprocess_run(monkeypatch) + _make_dummy_env(**kw) + return next(c[0] for c in calls if isinstance(c[0], list) and len(c[0]) >= 2 and c[0][1] == "run") + + default, compat = run_args(), run_args(snap_compat=True) + assert "--init" in default and "no-new-privileges" in default + assert "--init" not in compat and "no-new-privileges" not in compat + + def strip(argv): # everything except the two flags and the random container name + return [a for a in argv if a not in ("--init", "--security-opt", "no-new-privileges") and not a.startswith("hermes-")] + + assert strip(default) == strip(compat) + + # ── run_as_host_user tests ──────────────────────────────────────── diff --git a/tests/tools/test_terminal_config_env_sync.py b/tests/tools/test_terminal_config_env_sync.py index c7b98babb1..021c8e46c5 100644 --- a/tests/tools/test_terminal_config_env_sync.py +++ b/tests/tools/test_terminal_config_env_sync.py @@ -321,3 +321,12 @@ def test_docker_forward_env_is_bridged_everywhere(): assert "docker_forward_env" in _gateway_env_map_keys() assert "docker_forward_env" in _save_config_env_sync_keys() assert "TERMINAL_DOCKER_FORWARD_ENV" in _terminal_tool_env_var_names() + + +def test_docker_snap_compat_is_bridged_everywhere(): + """#9730: ``terminal.docker_snap_compat`` must reach the container on the CLI, gateway and + ``hermes config set`` paths, like every other docker_* key (see docker_extra_args above).""" + assert "docker_snap_compat" in _cli_env_map_keys() + assert "docker_snap_compat" in _gateway_env_map_keys() + assert "docker_snap_compat" in _save_config_env_sync_keys() + assert "TERMINAL_DOCKER_SNAP_COMPAT" in _terminal_tool_env_var_names() From 9c4c548cd555905563b146a6974932b0c5eb8a01 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:54:31 +0530 Subject: [PATCH 047/276] docs(terminal): document docker_snap_compat --- website/docs/user-guide/configuration.md | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 0872b83e15..4d21bc3df8 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -289,6 +289,7 @@ terminal: docker_image: "nikolaik/python-nodejs:python3.11-nodejs20" docker_mount_cwd_to_workspace: false # Mount launch dir into /workspace docker_run_as_host_user: false # See "Running container as host user" below + docker_snap_compat: false # See "Snap-packaged Docker (AppArmor)" below docker_forward_env: # Host env vars to forward into container - "GITHUB_TOKEN" docker_env: # Literal env vars to inject (KEY=value) @@ -377,6 +378,7 @@ Every key under `terminal:` has an env-var override of the form `TERMINAL_ Date: Sat, 5 Sep 2026 21:01:10 +0530 Subject: [PATCH 048/276] fix(docker): teardown workers of reaper-detached containers are drained at exit (#86317) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit DockerEnvironment.cleanup() runs docker stop + docker rm -f on a daemon thread and keeps the handle on the env. The idle reaper pops the env out of _active_environments BEFORE calling cleanup(), and the atexit drain only iterated that registry — so a detached env's worker was unreachable and died with the interpreter, leaving a stopped (or, if the exit came fast enough, still running) labeled container behind while the log said the environment was cleaned. Every teardown worker is now also recorded in a module-level set in docker.py; _atexit_cleanup joins that set after the registry pass via DockerEnvironment.wait_for_all_teardowns (re-snapshotting each pass, since the reaper can start a worker while the drain runs). Finished workers are dropped from the set on each drain so it cannot grow across a long gateway life. Mechanism from #86344 by @PRATHAMESH75; this is the slim redo — one shared set plus one static drain hooked into the existing terminal_tool atexit, instead of a second atexit registration in docker.py. --- tools/environments/docker.py | 26 ++++++++++++++++++++++++++ tools/terminal_tool.py | 5 +++++ 2 files changed, 31 insertions(+) diff --git a/tools/environments/docker.py b/tools/environments/docker.py index c53ca663cc..6ef7e61123 100644 --- a/tools/environments/docker.py +++ b/tools/environments/docker.py @@ -16,6 +16,7 @@ import shutil import subprocess import sys import threading +import time import uuid from pathlib import Path from typing import Optional @@ -291,6 +292,11 @@ _RUN_TMPFS_EXEC = "--tmpfs", "/run:rw,exec,nosuid,size=64m" # back. Skipped when --user is passed: the container already starts unprivileged. _PRIVDROP_CAP_ARGS = ["--cap-add", "SETUID", "--cap-add", "SETGID"] +# Every ``cleanup()`` worker, independent of terminal_tool's active-env registry (see +# ``wait_for_all_teardowns``). +_TEARDOWN_THREADS: set[threading.Thread] = set() +_TEARDOWN_LOCK = threading.Lock() + _S6_INIT_ENTRYPOINTS = ("/init", "/package/admin/s6-overlay/command/init") @@ -1035,6 +1041,8 @@ class DockerEnvironment(BaseEnvironment): logger.warning(fail_msg, log_id, e) t = threading.Thread(target=_do_cleanup, daemon=True, name=f"hermes-cleanup-{log_id}") + with _TEARDOWN_LOCK: + _TEARDOWN_THREADS.add(t) t.start() self._cleanup_thread = t self._container_id = None @@ -1044,6 +1052,24 @@ class DockerEnvironment(BaseEnvironment): if not self._persistent: self._remove_bind_dirs() + @staticmethod + def wait_for_all_teardowns(timeout: float = 15.0) -> bool: + """Join every in-flight teardown worker, including those of envs the idle reaper already + popped out of the active registry (their handles are otherwise unreachable at exit, so + ``docker rm`` died with the interpreter and left a stopped container behind — #86317). + Re-snapshots each pass: the reaper may start a worker while the drain runs.""" + deadline = time.monotonic() + timeout + while True: + with _TEARDOWN_LOCK: + _TEARDOWN_THREADS.difference_update({t for t in _TEARDOWN_THREADS if not t.is_alive()}) + pending = list(_TEARDOWN_THREADS) + if not pending: + return True + remaining = deadline - time.monotonic() + if remaining <= 0: + return False + pending[0].join(timeout=remaining) + def wait_for_cleanup(self, timeout: float = 30.0) -> bool: """Block up to *timeout* seconds for the cleanup thread (atexit hook). True if it finished or none was started, False on timeout.""" diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index 6aed41dc62..713c5bdf23 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -21,6 +21,7 @@ import/patch target): ``terminal_tool_config`` (TERMINAL_* reads, ``_quiet``), import json import logging import os +import sys import time import threading import atexit @@ -715,6 +716,10 @@ def _atexit_cleanup(): if wait_fn is not None: with _quiet("wait_for_cleanup raised on exit"): # never block shutdown on a bad backend wait_fn(timeout=15.0) + # Workers of envs the idle reaper already detached are not in the registry (#86317). + if "tools.environments.docker" in sys.modules: + with _quiet("teardown drain raised on exit"): + sys.modules["tools.environments.docker"].DockerEnvironment.wait_for_all_teardowns(timeout=15.0) atexit.register(_atexit_cleanup) From 9dd6634c5635321cf38840cc30e9b51226689128 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:01:10 +0530 Subject: [PATCH 049/276] test(docker): atexit drain joins the worker of an env already popped from the registry; finished workers do not accumulate Both red on origin/main. --- tests/tools/test_docker_teardown_drain.py | 66 +++++++++++++++++++++++ 1 file changed, 66 insertions(+) create mode 100644 tests/tools/test_docker_teardown_drain.py diff --git a/tests/tools/test_docker_teardown_drain.py b/tests/tools/test_docker_teardown_drain.py new file mode 100644 index 0000000000..5dd9a49fc5 --- /dev/null +++ b/tests/tools/test_docker_teardown_drain.py @@ -0,0 +1,66 @@ +"""#86317: a container-teardown worker must be joined at exit even after the idle reaper popped its +env out of ``_active_environments`` (the atexit drain only knew the registry, so ``docker rm`` for a +detached env died with the interpreter and left a stopped, labeled container behind).""" + +import subprocess +import threading + +import tools.environments.docker as docker_env +import tools.terminal_tool as terminal_tool + + +def _env_with_slow_teardown(monkeypatch, release: threading.Event, seen: list): + docker_env._cgroup_limits_ok = True + monkeypatch.setattr(docker_env, "find_docker", lambda: "/usr/bin/docker") + + def _run(cmd, **kwargs): + argv = list(cmd) + if argv[1] == "rm": + release.wait(5) # the slow part: interpreter used to exit before it finished + if argv[1] in ("stop", "rm"): + seen.append(argv[1]) + out = "fake-cid\n" if argv[1] == "run" else "" + return subprocess.CompletedProcess(argv, 0, stdout=out, stderr="") + + monkeypatch.setattr(docker_env.subprocess, "run", _run) + monkeypatch.setattr(docker_env, "run_capture", lambda cmd, **kw: _run(cmd)) + monkeypatch.setattr(docker_env.DockerEnvironment, "init_session", lambda self: None) + monkeypatch.setattr(docker_env.DockerEnvironment, "_remove_bind_dirs", lambda self: None) + return docker_env.DockerEnvironment( + image="python:3.11", cwd="/root", timeout=5, task_id="t-detach", persistent_filesystem=False, + persist_across_processes=False) + + +def test_atexit_drain_joins_worker_of_env_already_popped_from_registry(monkeypatch): + release, seen = threading.Event(), [] + env = _env_with_slow_teardown(monkeypatch, release, seen) + # The reaper's shape: pop first, then cleanup() — the registry no longer references env. + monkeypatch.setattr(terminal_tool, "_active_environments", {}) + env.cleanup() + assert env._cleanup_thread.is_alive() + + joined = {} + + def _drain(): + joined["result"] = None + terminal_tool._atexit_cleanup() + joined["result"] = True + + drainer = threading.Thread(target=_drain) + drainer.start() + drainer.join(0.5) + assert drainer.is_alive(), "drain returned while a detached teardown worker was still running" + release.set() + drainer.join(5) + assert joined["result"] is True and seen == ["stop", "rm"] + assert not env._cleanup_thread.is_alive() + + +def test_finished_workers_do_not_accumulate(): + t = threading.Thread(target=lambda: None) + with docker_env._TEARDOWN_LOCK: + docker_env._TEARDOWN_THREADS.add(t) + t.start(); t.join() + assert docker_env.DockerEnvironment.wait_for_all_teardowns(timeout=1) is True + with docker_env._TEARDOWN_LOCK: + assert t not in docker_env._TEARDOWN_THREADS From a8ca904922c122b2cc1d4955e472652815e1bc8a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 09:13:10 -0700 Subject: [PATCH 050/276] =?UTF-8?q?feat(evals):=20post-mortem=20harness=20?= =?UTF-8?q?=E2=80=94=20forensics=20lanes=20+=20live=20A/B=20+=20review=20p?= =?UTF-8?q?robes=20for=20the=20#102117=20run=20fixes?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit evals/postmortem/ turns the one-off audit behind tracking issue #103563 into something anyone with a Hermes state.db copy (and optionally rotated agent.log*) can run on their own fan-out: forensics/ common.py discovers the run tree (root = most descendants, compression-rollover children excluded so cost buckets stay disjoint), fits pricing from estimated_cost_usd, and five lanes recompute the OBSERVED figures: tokens (buckets, depth/duration shares, context reconstruction, excess-cache-write proxy, cap replay), logcalls (per-call cache behaviour from agent.log with coverage printed first; strict and loose plateau definitions reported separately), delegation (timeouts, orphaned children, polling hours, batch-join withheld child-hours, truncated summaries), tools (hardline blocks, foreground refusals, whole-file rewrites), goal_loop (nudges, parked barrier), rework (public-surface drop at PR open + post-open commit inventory). Every figure is labeled OBSERVED or MODELED. live_ab/ the per-PR A/Bs (real code paths, fake providers, temp HERMES_HOME), paths from argv. review_probes/ the independent /review's probes, credited and adapted; each reproduced a round-1 defect and the fixed head must pass it. run.py runs the offline probes against one or two checkouts and prints PASS/FAIL side by side (--live adds the ones that spend cents). tests/ synthetic-DB smoke test for the lanes and runner. On the run's DB the lanes reproduce the tracking issue's population exactly (1,394 sessions, 93,284 calls, $19,302.59; cache_write $11,159.76) and on main vs an integration checkout of the 13 PRs the runner shows every probe FAIL -> PASS (two guard-only probes pass on both, noted in run.py). The trajectories are deliberately not shipped: the DB holds 51,956 home paths, 5,341 e-mails, private IPs, chat ids and real-shaped credentials in tool output. The lane reports and recomputed JSON are in a secret gist linked from #103563. --- evals/postmortem/README.md | 117 +++++++++++ evals/postmortem/__init__.py | 0 evals/postmortem/forensics/__init__.py | 0 evals/postmortem/forensics/common.py | 186 ++++++++++++++++++ evals/postmortem/forensics/delegation.py | 115 +++++++++++ evals/postmortem/forensics/goal_loop.py | 70 +++++++ evals/postmortem/forensics/logcalls.py | 136 +++++++++++++ evals/postmortem/forensics/rework.py | 76 +++++++ evals/postmortem/forensics/tokens.py | 122 ++++++++++++ evals/postmortem/forensics/tools.py | 100 ++++++++++ evals/postmortem/live_ab/__init__.py | 0 evals/postmortem/live_ab/auth_stampede.py | 79 ++++++++ .../live_ab/batch_failure_notice.py | 42 ++++ evals/postmortem/live_ab/cache_prefix_live.py | 49 +++++ evals/postmortem/live_ab/cache_prefix_wire.py | 79 ++++++++ evals/postmortem/live_ab/goal_judge_wait.py | 19 ++ .../live_ab/hardline_scanner_matrix.py | 26 +++ .../live_ab/nested_delegate_deadline.py | 45 +++++ .../live_ab/subagent_context_cap.py | 32 +++ evals/postmortem/review_probes/__init__.py | 0 .../review_probes/cache_estimator_probe.py | 38 ++++ .../review_probes/context_cap_probe.py | 95 +++++++++ .../credential_identity_probe.py | 113 +++++++++++ .../review_probes/deadline_probe.py | 140 +++++++++++++ .../review_probes/finalizer_schedule_probe.py | 56 ++++++ .../review_probes/goal_repaste_probe.py | 129 ++++++++++++ .../review_probes/goal_scope_probe.py | 78 ++++++++ .../review_probes/notice_delivery_probe.py | 86 ++++++++ .../review_probes/rewrite_hint_probe.py | 48 +++++ .../review_probes/scanner_bypass_probe.py | 55 ++++++ evals/postmortem/run.py | 87 ++++++++ evals/postmortem/tests/__init__.py | 0 .../tests/test_postmortem_harness.py | 88 +++++++++ 33 files changed, 2306 insertions(+) create mode 100644 evals/postmortem/README.md create mode 100644 evals/postmortem/__init__.py create mode 100644 evals/postmortem/forensics/__init__.py create mode 100644 evals/postmortem/forensics/common.py create mode 100644 evals/postmortem/forensics/delegation.py create mode 100644 evals/postmortem/forensics/goal_loop.py create mode 100644 evals/postmortem/forensics/logcalls.py create mode 100644 evals/postmortem/forensics/rework.py create mode 100644 evals/postmortem/forensics/tokens.py create mode 100644 evals/postmortem/forensics/tools.py create mode 100644 evals/postmortem/live_ab/__init__.py create mode 100644 evals/postmortem/live_ab/auth_stampede.py create mode 100644 evals/postmortem/live_ab/batch_failure_notice.py create mode 100644 evals/postmortem/live_ab/cache_prefix_live.py create mode 100644 evals/postmortem/live_ab/cache_prefix_wire.py create mode 100644 evals/postmortem/live_ab/goal_judge_wait.py create mode 100644 evals/postmortem/live_ab/hardline_scanner_matrix.py create mode 100644 evals/postmortem/live_ab/nested_delegate_deadline.py create mode 100644 evals/postmortem/live_ab/subagent_context_cap.py create mode 100644 evals/postmortem/review_probes/__init__.py create mode 100644 evals/postmortem/review_probes/cache_estimator_probe.py create mode 100644 evals/postmortem/review_probes/context_cap_probe.py create mode 100644 evals/postmortem/review_probes/credential_identity_probe.py create mode 100644 evals/postmortem/review_probes/deadline_probe.py create mode 100644 evals/postmortem/review_probes/finalizer_schedule_probe.py create mode 100644 evals/postmortem/review_probes/goal_repaste_probe.py create mode 100644 evals/postmortem/review_probes/goal_scope_probe.py create mode 100644 evals/postmortem/review_probes/notice_delivery_probe.py create mode 100644 evals/postmortem/review_probes/rewrite_hint_probe.py create mode 100644 evals/postmortem/review_probes/scanner_bypass_probe.py create mode 100644 evals/postmortem/run.py create mode 100644 evals/postmortem/tests/__init__.py create mode 100644 evals/postmortem/tests/test_postmortem_harness.py diff --git a/evals/postmortem/README.md b/evals/postmortem/README.md new file mode 100644 index 0000000000..742f28dff3 --- /dev/null +++ b/evals/postmortem/README.md @@ -0,0 +1,117 @@ +# Post-mortem harness: forensics + live A/B for the 1,393-agent run fixes + +The scripts that produced every number in tracking issue #103563 and the "Independent review +(round 2)" sections of its 13 PRs. Two halves: + +- **`forensics/`** reads a *copy* of a Hermes `state.db` (plus rotated `agent.log*` and git) and + recomputes the *observed* figures for any run: where the money went, per-call cache behaviour, + nested-delegate timeouts, batch-join delivery delay, tool friction, `/goal` loop behaviour, and the + post-open rework inventory. It needs no model calls and no network. +- **`live_ab/`** and **`review_probes/`** exercise the real code paths of a checkout (real + `AIAgent`, dispatch, judge, scanner, SDK) against local fake providers to show what each fix + *does*. `run.py` runs them against one or two checkouts and prints PASS/FAIL side by side. + `review_probes/` are the probes the independent `/review` wrote; each reproduced a defect in the + first version of a PR and the fixed head must pass it. + +Everything is labeled **OBSERVED** (from usage rows / logs / git) or **MODELED** (a reconstruction or +replay). Do not add the modeled figures to the observed ones; see §5 of #103563 for why they overlap. + +## Requirements + +- A Hermes checkout with its venv (`.venv/bin/python`). NeMo Relay is not required for the forensics; + it is what the run itself used for the wire captures in `live_ab/cache_prefix_wire.py`. +- For forensics: a **copy** of `~/.hermes/state.db` (never point at the live file; `sqlite3 state.db ".backup copy.db"` + or `cp` while Hermes is idle) and, optionally, the rotated `~/.hermes/logs/agent.log*`. +- For `--live` probes: real credentials in `HERMES_HOME` (they spend cents per run). + +## Forensics: recompute the observed numbers for YOUR run + +```bash +cd +P=.venv/bin/python +# 1. cost buckets, depth/duration shares, context-size reconstruction, excess-cache-write proxy, cap replay +$P -m evals.postmortem.forensics.tokens --db state_copy.db [--root ] [--cap 200000 --floor 65000] +# 2. per-call cache behaviour from the logs (coverage fraction is printed first; quote nothing without it) +$P -m evals.postmortem.forensics.logcalls --db state_copy.db --logs "$HOME/.hermes/logs/agent.log*" +# 3. delegation: timeouts, orphaned children, polling, batch-join delay, truncated summaries +$P -m evals.postmortem.forensics.delegation --db state_copy.db +# 4. tool friction: hardline false blocks, foreground refusals, whole-file rewrites, output volume +$P -m evals.postmortem.forensics.tools --db state_copy.db +# 5. /goal loop: nudges, parked barrier, notification counts +$P -m evals.postmortem.forensics.goal_loop --db state_copy.db +# 6. rework inventory for a large PR (git only) +$P -m evals.postmortem.forensics.rework --repo . --base --open --head +``` + +Each writes `postmortem_out/.json` and prints a summary. `--root` defaults to the top-level +session with the most descendants; compression-rollover children are excluded from the population so +cost buckets are disjoint. Pricing is fitted from `estimated_cost_usd`, so dollars match what that +Hermes recorded (an estimator, not an invoice). + +### Reference output (the #102117 run, `state_copy.db` of 2026-09-04) + +| lane | prints | +|---|---| +| tokens | `1394 sessions, $19,302.59`; buckets cache_write $11,159.76 · cache_read $3,587.17 · output $4,555.48; depth-2 65%; >60 min 61%; MODELED excess cache-write proxy ~$9.1k; sawtooth cap 200K: prompt volume ×0.76 (reconstructed sizes) | +| logcalls | coverage 22,489/93,284 (24.1%); median prompt 229,648, p90 408,794, >200K 59%; hit ratio 93.9%; strict plateau 22.5% of uncached input, non-advancing 74.8%; sawtooth cap 200K on real sizes ×0.498 | +| delegation | 266 orchestrators; 332 delegate_task timeouts in 234 sessions (all nested); 93 results carried; 242.6 h sleep after first timeout; those sessions' lifetime $4,034.69 (includes real work); summaries truncated 123/220; batch-join withheld child-hours root 233 / depth-1 300 / depth-2 52 | +| tools | 96,855 tool calls; hardline blocks 579 (566 "malformed" class); foreground timeout refusals 475 (303 asked 900 s); `&` 185, nohup 24; write_file 8,188 calls / 92.8M chars, 661 rewrites of a file read this session >20k; patch 4,623 | +| goal_loop | 5 nudges (2 within 180 s of a "waiting" turn); 34 batch notices; 48 bg-process notices; final barrier `waiting_on_session=proc_…` parked 201 min | +| rework | surface at open: 1,703 names / 341 modules, 951 methods / 156, 126 test defs / 52 files; 125 post-open commits (simplify 55, review-fix 39, fix 19, …) | + +The two sawtooth figures differ on purpose: `tokens` replays *reconstructed* per-call sizes for all +93k calls (×0.76); `logcalls` replays *real* per-call sizes for the 24% of calls in the logs (×0.50, +the peak-concurrency window). The tracking issue quotes the second and says so. + +## Live A/B: what each fix does + +```bash +# offline probes (fake providers, temp HERMES_HOME), main vs a branch or integration checkout: +.venv/bin/python -m evals.postmortem.run --repo /path/to/main --compare /path/to/branch +# add the probes that make real provider calls (cents): +.venv/bin/python -m evals.postmortem.run --repo /path/to/branch --live +# one PR only: +.venv/bin/python -m evals.postmortem.run --repo . --only 103492 +``` + +| probe | PR | expects on the fixed head | +|---|---|---| +| `live_ab/hardline_scanner_matrix.py` | #103492 | 11-case matrix `ALL OK` (546-block class allowed; newline/`;`/`&&`/`|` hidden `reboot` blocked as itself) | +| `review_probes/scanner_bypass_probe.py` | #103492 | public guard `approved: False`, 0 callbacks, harmless Bash witness not executed | +| `live_ab/subagent_context_cap.py` | #103513 | child trigger `200,000` on a 1M model; parent untouched | +| `review_probes/context_cap_probe.py` | #103513 | cap holds through repeated compression + persistence; config validation | +| `live_ab/nested_delegate_deadline.py` | #103486 | 40 s deadline + 75 s leaf: result delivered (main: lost) | +| `review_probes/deadline_probe.py` | #103486 | same through actual dispatch | +| `live_ab/auth_stampede.py 40` | #103526 | `401s=0` (main: 40) | +| `review_probes/credential_identity_probe.py pr` | #103526 | explicit account-A key stays A (v1: became B) | +| `live_ab/batch_failure_notice.py` | #103549 | `TASK_FAILURE_NOTICE` at t+0.3 s, `BATCH_FINAL` after | +| `review_probes/notice_delivery_probe.py` | #103549 | gateway receives notice, notice, final; busy-parent final claim succeeds | +| `review_probes/cache_estimator_probe.py` | #103476 | preflight ≈ wire estimate; `should_compress` agrees | +| `live_ab/cache_prefix_wire.py B` (live) | #103476 | 0 mutated prefixes across 6 calls | +| `live_ab/goal_judge_wait.py 3` (live) | #103534 | `wait` ×3 on the run's "waiting on workers" response (main: `continue` ×3) | +| `review_probes/goal_scope_probe.py` | #103496/#103534 | judge sees own processes; delegation WAIT lifts on batch return | +| `review_probes/goal_repaste_probe.py` | #103553 | near-whole re-paste → pointer; `ship the API` ≠ `ship the UI` | +| `review_probes/rewrite_hint_probe.py` | #103551 | remote backend: no host-derived hint; FIFO: returns; 460 KB repeated-line file: skipped, not 22 s | +| `review_probes/finalizer_schedule_probe.py` | #103507 | pytest plugin: `-p evals.postmortem.review_probes.finalizer_schedule_probe --finalizer-probe=consumer-first` on `tests/e2e/test_relay_native_openai_stream.py` → 2 passed | + +## What is NOT here, and why + +- **The trajectories themselves.** The run's `state.db` contains 51,956 absolute home paths, 5,341 + e-mail addresses, private IPs, chat/user ids, and real-shaped API keys and JWTs in tool output. It + is not publishable, and this harness is written so it does not need to be: run it on your own DB. +- **Hand classification.** The root-cause classes of the 72 rework commits (dropped symbol vs semantic + drift vs compat fallout) were labeled by hand in the original audit; `forensics/rework.py` reproduces + the mechanical inventory that labeling started from and stops there. +- **A single aggregate saving.** By design. Each lane prints its own number with its own caveat. + +## Evidence bundle + +The original lane reports, the independent review, and the JSON this harness recomputes on the run's DB +are in a secret gist linked from #103563 (no trajectories; see "What is NOT here"). + +## Provenance + +Forensic lanes: five parallel Hermes subagents (2026-09-04), rewritten here to take `--db`/`--root` +instead of hard-coded paths. `live_ab/`: the primary agent's per-PR A/Bs. `review_probes/`: the +independent `/review` subagent's probes (2026-09-05), adapted to take paths from the command line; +their findings and the fixes are in each PR's "Independent review (round 2)" section. diff --git a/evals/postmortem/__init__.py b/evals/postmortem/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evals/postmortem/forensics/__init__.py b/evals/postmortem/forensics/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evals/postmortem/forensics/common.py b/evals/postmortem/forensics/common.py new file mode 100644 index 0000000000..6d3d8f17a1 --- /dev/null +++ b/evals/postmortem/forensics/common.py @@ -0,0 +1,186 @@ +"""Shared loaders for the post-mortem forensics: point at ANY Hermes ``state.db`` (a copy, never the live +file) and get the run tree, the in-run session set, fitted pricing and message iterators. + +Nothing here knows about a particular run. The root is discovered as the session with the most +descendants unless ``--root`` is given; compression-rollover children (a child whose ``id`` the parent's +``compaction`` metadata names as its continuation) are excluded from the tree so cost populations stay +disjoint. Pricing is fitted by least squares from ``sessions`` usage columns to ``estimated_cost_usd``, so +the recomputed dollars match what THAT Hermes recorded, not an invoice. + +Usage from a lane script:: + + from evals.postmortem.forensics.common import Run + run = Run.from_args() # --db, --root, --out + for sid in run.in_run: # ordered session ids + ... + run.write("q1_cost.json", data) +""" +from __future__ import annotations + +import argparse +import collections +import json +import sqlite3 +from dataclasses import dataclass, field +from pathlib import Path +from typing import Any, Dict, Iterable, Iterator, List, Optional + +USAGE_COLS = ("input_tokens", "cache_read_tokens", "cache_write_tokens", "output_tokens") + + +def _lstsq(rows: List[List[float]], y: List[float]) -> List[float]: + """Ordinary least squares without numpy (4 unknowns): normal equations solved by Gaussian elimination.""" + n = len(rows[0]) + ata = [[sum(r[i] * r[j] for r in rows) for j in range(n)] for i in range(n)] + atb = [sum(r[i] * yy for r, yy in zip(rows, y)) for i in range(n)] + m = [row[:] + [b] for row, b in zip(ata, atb)] + for c in range(n): + piv = max(range(c, n), key=lambda r: abs(m[r][c])) + m[c], m[piv] = m[piv], m[c] + if abs(m[c][c]) < 1e-12: + continue + for r in range(n): + if r != c: + f = m[r][c] / m[c][c] + m[r] = [a - f * b for a, b in zip(m[r], m[c])] + return [m[i][n] / m[i][i] if abs(m[i][i]) > 1e-12 else 0.0 for i in range(n)] + + +@dataclass +class Run: + db_path: Path + out_dir: Path + root: str + sessions: Dict[str, Dict[str, Any]] + depth: Dict[str, int] + in_run: List[str] # root + descendants, rollover excluded, dispatch order + price_per_token: Dict[str, float] # fitted: USD per token for each USAGE_COLS entry + _conn: sqlite3.Connection = field(repr=False) + + # ── construction ────────────────────────────────────────────────────────────────────────── + @classmethod + def parser(cls, description: str = "") -> argparse.ArgumentParser: + ap = argparse.ArgumentParser(description=description) + ap.add_argument("--db", required=True, help="path to a COPY of ~/.hermes/state.db") + ap.add_argument("--root", default=None, help="root session id (default: the session with the most descendants)") + ap.add_argument("--out", default="postmortem_out", help="directory for JSON/markdown outputs") + return ap + + @classmethod + def from_args(cls, argv: Optional[List[str]] = None, description: str = "") -> "Run": + a = cls.parser(description).parse_args(argv) + return cls.open(a.db, root=a.root, out=a.out) + + @classmethod + def open(cls, db: str, *, root: Optional[str] = None, out: str = "postmortem_out") -> "Run": # noqa: C901 + conn = sqlite3.connect(f"file:{db}?mode=ro", uri=True) + conn.row_factory = sqlite3.Row + sessions = {r["id"]: dict(r) for r in conn.execute("SELECT * FROM sessions")} + children: Dict[Optional[str], List[str]] = collections.defaultdict(list) + for sid, s in sessions.items(): + children[s.get("parent_session_id")].append(sid) + rollover = cls._rollover_ids(conn, sessions) + if root is None: + def size(sid: str) -> int: + n, stack = 0, [sid] + while stack: + cur = stack.pop() + for c in children[cur]: + if c not in rollover: + n += 1; stack.append(c) + return n + root = str(max((sid for sid in sessions if sessions[sid].get("parent_session_id") is None), key=size)) + depth: Dict[str, int] = {} + order: List[str] = [] + stack = [(root, 0)] + while stack: + sid, d = stack.pop() + depth[sid] = d; order.append(sid) + stack.extend((c, d + 1) for c in sorted(children[sid], key=lambda x: sessions[x].get("started_at") or 0, reverse=True) if c not in rollover) + order.sort(key=lambda s: sessions[s].get("started_at") or 0) + price = cls._fit_pricing([sessions[s] for s in order]) + out_dir = Path(out); out_dir.mkdir(parents=True, exist_ok=True) + return cls(Path(db), out_dir, root, sessions, depth, order, price, conn) + + @staticmethod + def _rollover_ids(conn: sqlite3.Connection, sessions: Dict[str, Dict[str, Any]]) -> set: + """Children created by in-place compression rollover, not by delegation: a child whose ``source`` + is a top-level surface (cli/tui/telegram/...), not ``subagent``, whose parent shares that source, + and which started within a few seconds of the parent ending. Their whole later lifetime belongs to + the continuing conversation, not to the fan-out, so they are kept out of the run population.""" + out = set() + for sid, s in sessions.items(): + pid = s.get("parent_session_id") + if not pid or pid not in sessions or (s.get("source") or "") == "subagent": + continue + parent = sessions[pid] + if (s.get("source") or "") != (parent.get("source") or ""): + continue + try: + gap = float(s.get("started_at") or 0) - float(parent.get("ended_at") or 0) + except (TypeError, ValueError): + continue + if -5.0 <= gap <= 5.0: + out.add(sid) + return out + + @staticmethod + def _fit_pricing(rows: Iterable[Dict[str, Any]]) -> Dict[str, float]: + X, y = [], [] + for s in rows: + cost = s.get("estimated_cost_usd") + if not cost: + continue + X.append([float(s.get(c) or 0) for c in USAGE_COLS]); y.append(float(cost)) + if len(X) < 8: + return {c: 0.0 for c in USAGE_COLS} + # Columns with negligible mass (e.g. input_tokens on cache-heavy Anthropic routes) make the + # normal equations ill-conditioned; fit only columns carrying >0.1% of all tokens. + mass = [sum(r[i] for r in X) for i in range(len(USAGE_COLS))] + keep = [i for i, m in enumerate(mass) if m > 0.001 * sum(mass)] + coef = _lstsq([[r[i] for i in keep] for r in X], y) + price = {c: 0.0 for c in USAGE_COLS} + for i, k in enumerate(keep): + price[USAGE_COLS[k]] = max(0.0, coef[i]) + return price + + # ── accessors ───────────────────────────────────────────────────────────────────────────── + def cost(self, sid: str) -> float: + return float(self.sessions[sid].get("estimated_cost_usd") or 0.0) + + def messages(self, sid: str, cols: str = "*") -> List[Dict[str, Any]]: + return [dict(r) for r in self._conn.execute(f"SELECT {cols} FROM messages WHERE session_id=? ORDER BY id", (sid,))] + + def iter_messages(self, sids: Iterable[str], cols: str = "*") -> Iterator[Dict[str, Any]]: + for sid in sids: + yield from self.messages(sid, cols) + + def system_prompt_len(self, sid: str) -> int: + h = self.sessions[sid].get("system_prompt_hash") + if not h: + return 0 + r = self._conn.execute("SELECT length(prompt) AS n FROM system_prompts WHERE hash=?", (h,)).fetchone() + return int(r["n"]) if r else 0 + + def by_depth(self) -> Dict[int, List[str]]: + out: Dict[int, List[str]] = collections.defaultdict(list) + for sid in self.in_run: + out[self.depth[sid]].append(sid) + return dict(out) + + def write(self, name: str, data: Any) -> Path: + p = self.out_dir / name + p.write_text(json.dumps(data, indent=1, default=str) if not name.endswith(".md") else str(data), encoding="utf-8") + return p + + def summary(self) -> Dict[str, Any]: + tot = {c: sum(float(self.sessions[s].get(c) or 0) for s in self.in_run) for c in USAGE_COLS} + return { + "root": self.root, "sessions": len(self.in_run), "children": len(self.in_run) - 1, + "by_depth": {d: len(v) for d, v in sorted(self.by_depth().items())}, + "api_calls": sum(int(self.sessions[s].get("api_call_count") or 0) for s in self.in_run), + "cost_usd": round(sum(self.cost(s) for s in self.in_run), 2), + "usage_tokens": tot, + "fitted_price_per_million": {c: round(p * 1e6, 4) for c, p in self.price_per_token.items()}, + "cost_by_bucket_usd": {c: round(tot[c] * self.price_per_token[c], 2) for c in USAGE_COLS}, + } diff --git a/evals/postmortem/forensics/delegation.py b/evals/postmortem/forensics/delegation.py new file mode 100644 index 0000000000..0ebe5beef5 --- /dev/null +++ b/evals/postmortem/forensics/delegation.py @@ -0,0 +1,115 @@ +"""Lane 2: delegation behaviour — nested-delegate timeouts, orphaned children, polling cost, batch-join +delivery delay, and summary truncation. All OBSERVED from ``messages`` in the run tree. + + python -m evals.postmortem.forensics.delegation --db state_copy.db + +Reproduces: how many ``delegate_task`` tool results were deadline timeouts and in how many sessions; +how many nested calls ever returned a result; what those sessions spent after their first timeout +(lifetime cost, with the caveat that it includes real work); total ``sleep N`` seconds; child results +withheld by batch joins (concurrent child-hours, NOT critical path); and how many parent-side summaries +carried the truncation footer. +""" +from __future__ import annotations + +import collections +import json +import re +import statistics +from typing import Any, Dict, List + +from evals.postmortem.forensics.common import Run + +_TIMEOUT = re.compile(r"timed out after ([\d.]+)s") +_SLEEP = re.compile(r"\bsleep\s+(\d+)") +_TRUNC = re.compile(r"\[SUMMARY TRUNCATED\]|middle omitted|trimmed to protect the parent") + + +def main(argv=None) -> int: + run = Run.from_args(argv, (__doc__ or "").split("\n\n")[0]) + parent_of = {s: run.sessions[s].get("parent_session_id") for s in run.in_run} + children_of: Dict[str, List[str]] = collections.defaultdict(list) + for s, p in parent_of.items(): + if p: + children_of[p].append(s) + orchestrators = [s for s in run.in_run if children_of.get(s)] + + timeouts_by_sess: Dict[str, int] = collections.Counter() + delegate_results = delegate_ok = 0 + sleep_seconds_after_timeout = 0 + first_timeout_ts: Dict[str, float] = {} + truncated_summaries = total_summaries = 0 + for sid in orchestrators: + for m in run.messages(sid, "role, tool_name, content, tool_calls, timestamp"): + if m["role"] == "tool" and (m.get("tool_name") or "") == "delegate_task": + c = m.get("content") or "" + delegate_results += 1 + if _TIMEOUT.search(c) and "delegate_task" in c: + timeouts_by_sess[sid] += 1 + first_timeout_ts.setdefault(sid, float(m.get("timestamp") or 0)) + elif '"status"' in c and ("completed" in c or "summary" in c): + delegate_ok += 1 + total_summaries += 1 + if _TRUNC.search(c): + truncated_summaries += 1 + if m["role"] == "user" and (m.get("content") or "").startswith("[ASYNC DELEGATION"): + # background batches re-enter as a user-role completion block, one entry per child + c = m["content"] + entries = c.count("--- ✓ TASK") + c.count("--- ✗ TASK") + c.count("--- ⚠ TASK") or (1 if "COMPLETE" in c else 0) + total_summaries += entries + truncated_summaries += c.count("[SUMMARY TRUNCATED]") + if m["role"] == "assistant" and sid in first_timeout_ts and float(m.get("timestamp") or 0) >= first_timeout_ts[sid]: + for mt in _SLEEP.finditer(m.get("tool_calls") or ""): + sleep_seconds_after_timeout += int(mt.group(1)) + + timeout_sessions = list(timeouts_by_sess) + nested_timeout_sessions = [s for s in timeout_sessions if run.depth[s] >= 1] + lifetime_cost = sum(run.cost(s) for s in timeout_sessions) + lifetime_cache_write = sum(float(run.sessions[s].get("cache_write_tokens") or 0) for s in timeout_sessions) * run.price_per_token["cache_write_tokens"] + + # batch-join delivery delay: children of one parent dispatched within 60 s of each other = one batch + withheld: Dict[int, List[float]] = collections.defaultdict(list) + for p, kids in children_of.items(): + kids = sorted(kids, key=lambda k: float(run.sessions[k].get("started_at") or 0)) + batch: List[str] = [] + def flush(batch: List[str]) -> None: + if len(batch) < 2: + return + ends = [float(run.sessions[k].get("ended_at") or 0) for k in batch] + last = max(ends) + withheld[run.depth.get(str(p), 0)].extend(last - e for e in ends if e) + for k in kids: + if batch and float(run.sessions[k].get("started_at") or 0) - float(run.sessions[batch[-1]].get("started_at") or 0) > 60: + flush(batch); batch = [] + batch.append(k) + flush(batch) + + report: Dict[str, Any] = { + "observed": { + "orchestrators": len(orchestrators), + "delegate_task_results": delegate_results, + "delegate_task_timeouts": sum(timeouts_by_sess.values()), + "sessions_with_timeouts": len(timeout_sessions), + "nested_sessions_with_timeouts": len(nested_timeout_sessions), + "delegate_task_results_that_carried_a_result": delegate_ok, + "sleep_hours_after_first_timeout": round(sleep_seconds_after_timeout / 3600, 1), + "timeout_sessions_lifetime_cost_usd": round(lifetime_cost, 2), + "timeout_sessions_lifetime_cache_write_usd": round(lifetime_cache_write, 2), + "lifetime_cost_caveat": "whole-session spend; includes legitimate work after the timeout, and its cache writes overlap the tokens lane's excess proxy", + "summaries_truncated": truncated_summaries, "summaries_total": total_summaries, + "batch_join_withheld_child_hours_by_parent_depth": {d: round(sum(v) / 3600, 1) for d, v in sorted(withheld.items())}, + "batch_join_withheld_minutes_median_by_parent_depth": {d: round(statistics.median(v) / 60, 1) for d, v in sorted(withheld.items()) if v}, + "withheld_caveat": "concurrent child-hours held back from the parent; not critical-path time", + } + } + path = run.write("delegation.json", report) + o = report["observed"] + print(f"[delegation] {o['orchestrators']} orchestrators; delegate_task results {o['delegate_task_results']:,}: timeouts {o['delegate_task_timeouts']} in {o['sessions_with_timeouts']} sessions " + f"({o['nested_sessions_with_timeouts']} nested); carried a result: {o['delegate_task_results_that_carried_a_result']}") + print(f"[delegation] after first timeout: sleep {o['sleep_hours_after_first_timeout']} h; those sessions' lifetime ${o['timeout_sessions_lifetime_cost_usd']:,} (cache writes ${o['timeout_sessions_lifetime_cache_write_usd']:,}; includes real work)") + print(f"[delegation] summaries truncated {o['summaries_truncated']}/{o['summaries_total']}; batch-join withheld child-hours by parent depth {o['batch_join_withheld_child_hours_by_parent_depth']}") + print(f"[delegation] wrote {path}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/evals/postmortem/forensics/goal_loop.py b/evals/postmortem/forensics/goal_loop.py new file mode 100644 index 0000000000..87200f1ab7 --- /dev/null +++ b/evals/postmortem/forensics/goal_loop.py @@ -0,0 +1,70 @@ +"""Lane 4: /goal loop behaviour — nudges, judge verdicts, parked time. OBSERVED from the root session's +messages and the persisted goal state. + + python -m evals.postmortem.forensics.goal_loop --db state_copy.db + +Reproduces: how many ``[Continuing toward your standing goal]`` nudges fired and how soon after the +previous assistant turn; how many of those turns had just said they were waiting; the goal's final +persisted wait barrier (what it was parked on, since when); and the count of async-batch and +background-process notifications that re-entered the root. +""" +from __future__ import annotations + +import json +import re +from typing import Any, Dict, List + +from evals.postmortem.forensics.common import Run + +_WAITING = re.compile(r"\b(waiting|wait(s)? on|nothing (else |new )?to dispatch|no action needed|standing by|until .* (finish|return|complete)|in flight|still running)\b", re.I) + + +def main(argv=None) -> int: + run = Run.from_args(argv, (__doc__ or "").split("\n\n")[0]) + root = run.root + msgs = run.messages(root, "role, content, timestamp") + nudges: List[Dict[str, Any]] = [] + batch_notices = bgproc_notices = 0 + last_assistant = None + for m in msgs: + c = m.get("content") or "" + if m["role"] == "assistant": + last_assistant = m + elif m["role"] == "user": + if c.startswith("[Continuing toward your standing goal]"): + gap = float(m["timestamp"]) - float(last_assistant["timestamp"]) if last_assistant else None + nudges.append({"ts": m["timestamp"], "gap_s": round(gap, 1) if gap is not None else None, + "prev_turn_said_waiting": bool(last_assistant and _WAITING.search(last_assistant.get("content") or ""))}) + elif c.startswith("[ASYNC DELEGATION"): + batch_notices += 1 + elif c.startswith("[IMPORTANT: Background process"): + bgproc_notices += 1 + goal_state: Dict[str, Any] = {} + try: + row = run._conn.execute("SELECT value FROM state_meta WHERE key=?", (f"goal:{root}",)).fetchone() + if row: + goal_state = json.loads(row[0]) + except Exception: + pass + parked = {k: goal_state.get(k) for k in ("status", "waiting_on_pid", "waiting_on_session", "waiting_until", "waiting_since", "waiting_reason", "last_verdict", "turns_used")} + if goal_state.get("waiting_since"): + # The barrier is never auto-cleared unless a turn re-evaluates it, so its age at session end is + # the time the loop sat parked (the state may have been cleared by the user afterwards). + end = float(run.sessions[root].get("ended_at") or 0) + parked["parked_minutes_until_session_end"] = round((end - float(goal_state["waiting_since"])) / 60, 1) if end else None + report = {"observed": { + "nudges": len(nudges), "nudges_within_180s_of_a_waiting_turn": sum(1 for n in nudges if n["prev_turn_said_waiting"] and (n["gap_s"] or 1e9) < 180), + "nudge_detail": nudges, "async_batch_notices": batch_notices, "background_process_notices": bgproc_notices, + "final_goal_state": parked, + }} + path = run.write("goal_loop.json", report) + o = report["observed"] + print(f"[goal] nudges {o['nudges']}, of which {o['nudges_within_180s_of_a_waiting_turn']} fired <180 s after a turn that said it was waiting; " + f"batch notices {o['async_batch_notices']}, bg-process notices {o['background_process_notices']}") + print(f"[goal] final goal state: {o['final_goal_state']}") + print(f"[goal] wrote {path}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/evals/postmortem/forensics/logcalls.py b/evals/postmortem/forensics/logcalls.py new file mode 100644 index 0000000000..c8457bd9c3 --- /dev/null +++ b/evals/postmortem/forensics/logcalls.py @@ -0,0 +1,136 @@ +"""Lane 1b: per-call cache behaviour from ``agent.log`` (OBSERVED provider usage per call). + +Hermes logs one line per API call:: + + ... INFO [] agent.conversation_loop: API call #N: model=... in= out= total=... latency=..s cache=/ + +Given the rotated logs, this reproduces: prompt-size distribution, cache hit-ratio buckets, the +"plateau" signature of a broken cache prefix (hit count stuck at the previous call's breakpoint while +``in`` grows), the share of uncached input those plateaus explain, and the sawtooth replay of an absolute +context cap on REAL per-call prompt sizes (the figure the tokens lane reported as -49%). + + python -m evals.postmortem.forensics.logcalls --db state_copy.db --logs ~/.hermes/logs/agent.log* + +Coverage caveat: logs rotate; report the fraction of the run's calls that were found before quoting +anything from this lane, and treat extrapolations as upper bounds. +""" +from __future__ import annotations + +import collections +import glob +import re +import statistics +from typing import Any, Dict, List + +from evals.postmortem.forensics.common import Run + +_LINE = re.compile( + r"^(\S+ \S+) INFO \[(\S+)\] agent\.conversation_loop: API call #(\d+): model=(\S+) provider=\S+ " + r"in=(\d+) out=(\d+) total=\d+ latency=([\d.]+)s cache=(\d+)/(\d+)" +) + + +def parse_logs(paths: List[str], sids: set) -> List[Dict[str, Any]]: + calls: List[Dict[str, Any]] = [] + for path in paths: + try: + fh = open(path, encoding="utf-8", errors="replace") + except OSError: + continue + with fh: + for line in fh: + m = _LINE.match(line) + if m and m.group(2) in sids: + calls.append({"ts": m.group(1), "sid": m.group(2), "n": int(m.group(3)), "model": m.group(4), + "inp": int(m.group(5)), "out": int(m.group(6)), "lat": float(m.group(7)), + "hit": int(m.group(8))}) + return calls + + +def sawtooth_real(by_sid: Dict[str, List[Dict[str, Any]]], cap: int, floor: int) -> float: + """Replay on REAL prompt sizes: appended = in[i] - in[i-1]; compress to floor when above cap.""" + total = 0.0 + for calls in by_sid.values(): + calls = sorted(calls, key=lambda c: c["n"]) + ctx = None + for i, c in enumerate(calls): + appended = c["inp"] if i == 0 else max(0, c["inp"] - calls[i - 1]["inp"]) + ctx = c["inp"] if ctx is None else ctx + appended + if ctx > cap: + ctx = float(floor) + total += ctx + return total + + +def main(argv=None) -> int: + ap = Run.parser((__doc__ or "").split("\n\n")[0]) + ap.add_argument("--logs", nargs="+", required=True, help="agent.log files (globs ok)") + ap.add_argument("--cap", type=int, default=200_000) + ap.add_argument("--floor", type=int, default=65_000) + a = ap.parse_args(argv) + run = Run.open(a.db, root=a.root, out=a.out) + paths = sorted(p for g in a.logs for p in glob.glob(g)) + calls = parse_logs(paths, set(run.in_run)) + total_calls = run.summary()["api_calls"] + if not calls: + print("[logcalls] no matching API-call lines found in", paths); return 1 + inp = [c["inp"] for c in calls] + hit_total, in_total = sum(c["hit"] for c in calls), sum(inp) + buckets = collections.Counter() + for c in calls: + r = c["hit"] / c["inp"] if c["inp"] else 0 + buckets["<50%" if r < .5 else "50-90%" if r < .9 else "90-97%" if r < .97 else "97-99%" if r < .99 else ">=99%"] += 1 + by_sid: Dict[str, List[Dict[str, Any]]] = collections.defaultdict(list) + for c in calls: + by_sid[c["sid"]].append(c) + # Two definitions, reported separately (an independent review caught them being conflated): + # strict plateau: hit count EXACTLY equal to the previous call's (prefix stuck at the old breakpoint) + # non-advancing: hit count did not grow while input did (includes partial misses of other causes) + strict_pairs = strict_uncached = loose_pairs = loose_uncached = pairs = uncached_total = 0 + for cs in by_sid.values(): + cs.sort(key=lambda c: c["n"]) + for prev, cur in zip(cs, cs[1:]): + pairs += 1 + u = max(0, cur["inp"] - cur["hit"]); uncached_total += u + if cur["inp"] > prev["inp"]: + if cur["hit"] == prev["hit"]: + strict_pairs += 1; strict_uncached += u + if cur["hit"] <= prev["hit"]: + loose_pairs += 1; loose_uncached += u + price = run.price_per_token + real = sum(inp); capped = sawtooth_real(by_sid, a.cap, a.floor) + report = { + "coverage": {"calls_found": len(calls), "run_api_calls": total_calls, "fraction": round(len(calls) / total_calls, 4) if total_calls else None, + "first_ts": min(c["ts"] for c in calls), "last_ts": max(c["ts"] for c in calls), + "note": "rotated logs; extrapolations from this window are upper bounds"}, + "observed": { + "prompt_tokens": {"median": int(statistics.median(inp)), "p90": sorted(inp)[int(.9 * len(inp))], "max": max(inp)}, + "share_of_calls_above": {t: round(sum(1 for x in inp if x > t) / len(inp), 3) for t in (150_000, 200_000, 300_000)}, + "cache_hit_ratio_overall": round(hit_total / in_total, 4), + "hit_ratio_buckets": dict(buckets), + "strict_plateau": {"pairs_share": round(strict_pairs / pairs, 4) if pairs else None, + "share_of_uncached_input": round(strict_uncached / uncached_total, 4) if uncached_total else None}, + "non_advancing_hit": {"pairs_share": round(loose_pairs / pairs, 4) if pairs else None, + "share_of_uncached_input": round(loose_uncached / uncached_total, 4) if uncached_total else None}, + "uncached_input_usd_in_window": round(uncached_total * price["cache_write_tokens"], 2), + }, + "modeled": { + "sawtooth_on_real_prompt_sizes": { + "cap": a.cap, "floor": a.floor, "prompt_tokens_ratio": round(capped / real, 3) if real else None, + "caveats": ["compression-call cost excluded", "read/write ratio assumed unchanged", "window only"], + } + }, + } + path = run.write("logcalls.json", report) + o = report["observed"] + print(f"[logcalls] coverage {len(calls):,}/{total_calls:,} calls ({report['coverage']['fraction']:.1%}) from {len(paths)} file(s)") + print(f"[logcalls] OBSERVED median prompt {o['prompt_tokens']['median']:,} p90 {o['prompt_tokens']['p90']:,}; >200K: {o['share_of_calls_above'][200000]:.0%}; hit ratio {o['cache_hit_ratio_overall']:.1%}") + print(f"[logcalls] OBSERVED strict plateau (hit unchanged): {o['strict_plateau']['pairs_share']:.1%} of pairs, {o['strict_plateau']['share_of_uncached_input']:.1%} of uncached input; " + f"non-advancing hit: {o['non_advancing_hit']['pairs_share']:.1%} / {o['non_advancing_hit']['share_of_uncached_input']:.1%}; uncached in window ${o['uncached_input_usd_in_window']:,}") + print(f"[logcalls] MODELED sawtooth cap {a.cap}/{a.floor} on real sizes: prompt volume x{report['modeled']['sawtooth_on_real_prompt_sizes']['prompt_tokens_ratio']}") + print(f"[logcalls] wrote {path}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/evals/postmortem/forensics/rework.py b/evals/postmortem/forensics/rework.py new file mode 100644 index 0000000000..612fb8acb8 --- /dev/null +++ b/evals/postmortem/forensics/rework.py @@ -0,0 +1,76 @@ +"""Lane 5: post-open rework inventory for a large PR — reproducible part only. + + python -m evals.postmortem.forensics.rework --repo . --base --open --head + +Reports (OBSERVED from git): the public-surface drop at PR open (via ``scripts/ci/check_public_surface.py``), +then every commit between the opening SHA and the merged head grouped by subject prefix +(``fix``, ``review-fix``, ``revert``, ``test``, ``docs``, ``simplify``, other) with files/insertions/deletions +and whether tests were touched. Root-cause classification of each commit (dropped symbol vs semantic +drift vs compat fallout ...) was done BY HAND in the original audit and is not reproducible here; the +counts this script prints are the mechanical inventory that classification started from. +""" +from __future__ import annotations + +import argparse +import collections +import json +import re +import subprocess +import sys +from pathlib import Path + + +def _git(repo: str, *args: str) -> str: + return subprocess.run(["git", "-C", repo, *args], capture_output=True, text=True, encoding="utf-8", errors="replace").stdout + + +def main(argv=None) -> int: + ap = argparse.ArgumentParser(description=(__doc__ or "").split("\n\n")[0]) + ap.add_argument("--repo", default=".") + ap.add_argument("--base", required=True, help="merge-base of the PR") + ap.add_argument("--open", required=True, help="head SHA when the PR was opened / first reviewed") + ap.add_argument("--head", required=True, help="final merged SHA") + ap.add_argument("--out", default="postmortem_out") + a = ap.parse_args(argv) + out = Path(a.out); out.mkdir(parents=True, exist_ok=True) + + # scripts/ci/check_public_surface.py ships with #103541; look for it in this tree, then the target repo. + candidates = [Path(__file__).resolve().parents[3] / "scripts" / "ci" / "check_public_surface.py", + Path(a.repo).resolve() / "scripts" / "ci" / "check_public_surface.py"] + checker = next((c for c in candidates if c.exists()), None) + if checker is None: + surface_line = "(scripts/ci/check_public_surface.py not found; merge #103541 or pass a checkout that has it)" + else: + surface = subprocess.run([sys.executable, str(checker), "--base", a.base, "--head", a.open, "--json", str(out / "surface_at_open.json")], + cwd=a.repo, capture_output=True, text=True, encoding="utf-8", errors="replace") + surface_line = (surface.stdout.splitlines() or [f"(check_public_surface failed: {surface.stderr.strip()[:120]})"])[0] + + log = _git(a.repo, "log", "--no-merges", "--format=%H%x00%s", f"{a.open}..{a.head}") + groups = collections.defaultdict(list) + for line in log.splitlines(): + if not line.strip(): + continue + sha, subj = line.split("\x00", 1) + m = re.match(r"^([a-z-]+)(\(|:|!)", subj) + known = {"fix", "review-fix", "revert", "test", "docs", "simplify", "feat", "chore", "ci"} + prefix = m.group(1) if m and m.group(1) in known else "other" + files = _git(a.repo, "show", "--format=", "--name-only", sha).split() + stat = _git(a.repo, "show", "--format=", "--shortstat", sha) + ins = int((re.search(r"(\d+) insertion", stat) or [0, 0])[1]); dele = int((re.search(r"(\d+) deletion", stat) or [0, 0])[1]) + tests = [f for f in files if f.startswith("tests/") or "/tests/" in f] + groups[prefix].append({"sha": sha[:10], "subject": subj[:120], "files": len(files), "ins": ins, "del": dele, + "test_files": len(tests), "src_only": len(tests) == 0 and len(files) > 0}) + summary = {g: {"commits": len(v), "files": sum(x["files"] for x in v), "ins": sum(x["ins"] for x in v), "del": sum(x["del"] for x in v), + "touching_no_tests": sum(1 for x in v if x["src_only"])} for g, v in groups.items()} + report = {"surface_at_open": surface_line, "post_open_commits_by_prefix": summary, "commits": groups, + "note": "root-cause classes were hand-labeled in the original audit; not reproduced here"} + (out / "rework.json").write_text(json.dumps(report, indent=1), encoding="utf-8") + print(f"[rework] surface at open: {surface_line}") + total = sum(v["commits"] for v in summary.values()) + print(f"[rework] {total} post-open commits by prefix: " + ", ".join(f"{g} {v['commits']} ({v['touching_no_tests']} w/o tests)" for g, v in sorted(summary.items(), key=lambda kv: -kv[1]['commits']))) + print(f"[rework] wrote {out / 'rework.json'}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/evals/postmortem/forensics/tokens.py b/evals/postmortem/forensics/tokens.py new file mode 100644 index 0000000000..2312ed0238 --- /dev/null +++ b/evals/postmortem/forensics/tokens.py @@ -0,0 +1,122 @@ +"""Lane 1: where the money went, context sizes, and the compression-cap counterfactual. + +Reproduces (for any run): cost by bucket and depth, the share of calls above context thresholds, +the excess-cache-write proxy, and the sawtooth replay of an absolute subagent context cap. + + python -m evals.postmortem.forensics.tokens --db state_copy.db [--cap 200000 --floor 65000] + +Every figure is labeled in the output as OBSERVED (from usage rows) or MODELED (reconstruction / +replay). The per-call context reconstruction estimates tokens from message character counts +(chars/3.5) plus the system-prompt and tool-schema sizes; it is a diagnostic, not a measurement. +""" +from __future__ import annotations + +import collections +import statistics +from typing import Any, Dict, List + +from evals.postmortem.forensics.common import Run + +CHARS_PER_TOKEN = 3.5 +TOOL_SCHEMA_TOKENS = 12_000 + + +def _per_call_context(run: Run, sid: str) -> List[Dict[str, Any]]: + """Reconstructed prompt size at each assistant turn (MODELED).""" + ctx = run.system_prompt_len(sid) / CHARS_PER_TOKEN + TOOL_SCHEMA_TOKENS + calls, appended = [], 0.0 + for m in run.messages(sid, "role, length(coalesce(content,'')) AS lc, length(coalesce(tool_calls,'')) AS ltc"): + if m["role"] == "assistant": + calls.append({"ctx": ctx, "appended_since_last": appended}) + out = (m["lc"] + m["ltc"]) / CHARS_PER_TOKEN + ctx += out; appended = out + else: + ctx += m["lc"] / CHARS_PER_TOKEN; appended += m["lc"] / CHARS_PER_TOKEN + return calls + + +def sawtooth(calls: List[Dict[str, Any]], cap: int, floor: int) -> float: + """Prompt tokens if the session compressed to ``floor`` whenever ctx exceeded ``cap`` (MODELED).""" + total, ctx = 0.0, None + for c in calls: + ctx = c["ctx"] if ctx is None else ctx + c["appended_since_last"] + if ctx > cap: + ctx = float(floor) + total += ctx + return total + + +def main(argv=None) -> int: + ap = Run.parser((__doc__ or "").split("\n\n")[0]) + ap.add_argument("--cap", type=int, default=200_000) + ap.add_argument("--floor", type=int, default=65_000) + a = ap.parse_args(argv) + run = Run.open(a.db, root=a.root, out=a.out) + price = run.price_per_token + report: Dict[str, Any] = {"summary": run.summary()} + + # OBSERVED: cost by depth, by duration bucket, top-N concentration + by_depth = collections.defaultdict(float) + for sid in run.in_run: + by_depth[run.depth[sid]] += run.cost(sid) + costs = sorted((run.cost(s) for s in run.in_run), reverse=True) + total = sum(costs) or 1.0 + long_sessions = [s for s in run.in_run if (float(run.sessions[s].get("ended_at") or 0) - float(run.sessions[s].get("started_at") or 0)) > 3600] + report["observed"] = { + "cost_by_depth_usd": {d: round(v, 2) for d, v in sorted(by_depth.items())}, + "top30_share": round(sum(costs[:30]) / total, 3), + "sessions_over_60min": len(long_sessions), + "sessions_over_60min_cost_share": round(sum(run.cost(s) for s in long_sessions) / total, 3), + } + + # MODELED: per-call context distribution, excess cache-write proxy, sawtooth replay + all_calls, per_session = [], {} + est_total = actual_total = appended_total = 0.0 + for sid in run.in_run: + calls = _per_call_context(run, sid) + if not calls: + continue + per_session[sid] = calls + all_calls.extend(c["ctx"] for c in calls) + appended_total += sum(c["appended_since_last"] for c in calls) + calls[0]["ctx"] + est_total += sum(c["ctx"] for c in calls) + s = run.sessions[sid] + actual_total += float(s.get("cache_read_tokens") or 0) + float(s.get("cache_write_tokens") or 0) + float(s.get("input_tokens") or 0) + n = len(all_calls) or 1 + thresholds = {t: round(sum(1 for c in all_calls if c > t) / n, 3) for t in (100_000, 150_000, 200_000, 300_000)} + cache_write_tokens = sum(float(run.sessions[s].get("cache_write_tokens") or 0) for s in run.in_run) + excess = max(0.0, cache_write_tokens - appended_total) + base_prompt = sum(sum(c["ctx"] for c in v) for v in per_session.values()) + capped_prompt = sum(sawtooth(v, a.cap, a.floor) for v in per_session.values()) + ctx_spend = sum(float(run.sessions[s].get(c) or 0) * price[c] for s in run.in_run for c in ("cache_read_tokens", "cache_write_tokens")) + report["modeled"] = { + "note": "reconstructed from message sizes at chars/3.5; diagnostic, not a measurement", + "calls_reconstructed": len(all_calls), + "median_prompt_tokens": int(statistics.median(all_calls)) if all_calls else 0, + "share_of_calls_above": thresholds, + "reconstruction_vs_actual_ratio": round(est_total / actual_total, 3) if actual_total else None, + "excess_cache_write_proxy": { + "ideal_tokens_if_only_appended": int(appended_total), "actual_cache_write_tokens": int(cache_write_tokens), + "excess_tokens": int(excess), "excess_usd": round(excess * price["cache_write_tokens"], 2), + "caveats": ["token counts estimated", "ideal omits initial system/tool prefix writes", "reasoning accumulation omitted"], + }, + "sawtooth_replay": { + "cap": a.cap, "floor": a.floor, + "prompt_tokens_ratio_capped_vs_actual": round(capped_prompt / base_prompt, 3) if base_prompt else None, + "context_spend_usd_observed": round(ctx_spend, 2), + "context_spend_usd_if_capped": round(ctx_spend * capped_prompt / base_prompt, 2) if base_prompt else None, + "caveats": ["excludes compression-call cost", "assumes read/write ratio unchanged", "ignores re-reads and quality effects"], + }, + } + path = run.write("tokens.json", report) + o, m = report["observed"], report["modeled"] + print(f"[tokens] {run.summary()['sessions']} sessions, ${run.summary()['cost_usd']:,} | buckets {run.summary()['cost_by_bucket_usd']}") + print(f"[tokens] OBSERVED depth-2 share {by_depth.get(2,0)/total:.0%}, >60min share {o['sessions_over_60min_cost_share']:.0%}, top30 {o['top30_share']:.1%}") + print(f"[tokens] MODELED calls >150K ctx {m['share_of_calls_above'][150000]:.0%}; excess cache-write proxy ${m['excess_cache_write_proxy']['excess_usd']:,}") + print(f"[tokens] MODELED sawtooth cap {a.cap}: prompt volume x{m['sawtooth_replay']['prompt_tokens_ratio_capped_vs_actual']} -> context spend ${m['sawtooth_replay']['context_spend_usd_observed']:,} -> ${m['sawtooth_replay']['context_spend_usd_if_capped']:,}") + print(f"[tokens] wrote {path}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/evals/postmortem/forensics/tools.py b/evals/postmortem/forensics/tools.py new file mode 100644 index 0000000000..f824bc2fed --- /dev/null +++ b/evals/postmortem/forensics/tools.py @@ -0,0 +1,100 @@ +"""Lane 3: tool-layer friction — hardline false blocks, foreground refusals, whole-file rewrites, output +volume by tool. All OBSERVED from ``messages`` (role='tool') in the run tree. + + python -m evals.postmortem.forensics.tools --db state_copy.db +""" +from __future__ import annotations + +import collections +import json +import re +from typing import Any, Dict + +from evals.postmortem.forensics.common import Run + +_HARDLINE = "BLOCKED (hardline)" +_MALFORMED = "command parser limit or malformed executable payload" +_FG_TIMEOUT = re.compile(r"Foreground timeout (\d+)s exceeds the maximum") +_FG_AMP = "Foreground command uses '&' backgrounding" +_FG_WRAP = "shell-level background wrappers" +_DEADLINE = re.compile(r"timed out after ([\d.]+)s") + + +def main(argv=None) -> int: + run = Run.from_args(argv, (__doc__ or "").split("\n\n")[0]) + by_tool_bytes: Dict[str, int] = collections.Counter() + by_tool_calls: Dict[str, int] = collections.Counter() + hardline = malformed = fg_timeout = fg_amp = fg_wrap = tool_deadline = 0 + fg_timeout_requested: Dict[int, int] = collections.Counter() + write_calls = big_writes = rewrite_of_existing = 0 + write_chars = 0 + read_paths_by_sess: Dict[str, set] = collections.defaultdict(set) + for sid in run.in_run: + for m in run.messages(sid, "role, tool_name, content, tool_calls"): + if m["role"] == "assistant" and m.get("tool_calls"): + try: + tcs = json.loads(m["tool_calls"]) + except ValueError: + continue + for tc in tcs: + fn = (tc.get("function") or {}) + name = fn.get("name") or "" + by_tool_calls[name] += 1 + if name in ("read_file", "write_file", "patch"): + try: + args = json.loads(fn.get("arguments") or "{}") + except ValueError: + args = {} + if name == "read_file" and args.get("path"): + read_paths_by_sess[sid].add(args["path"]) + if name == "write_file": + write_calls += 1 + content = args.get("content") or "" + write_chars += len(content) + if len(content) > 20_000: + big_writes += 1 + if args.get("path") in read_paths_by_sess[sid]: + rewrite_of_existing += 1 + continue + if m["role"] != "tool": + continue + c = m.get("content") or "" + by_tool_bytes[m.get("tool_name") or "?"] += len(c) + if _HARDLINE in c: + hardline += 1 + if _MALFORMED in c: + malformed += 1 + mt = _FG_TIMEOUT.search(c) + if mt: + fg_timeout += 1; fg_timeout_requested[int(mt.group(1))] += 1 + if _FG_AMP in c: + fg_amp += 1 + if _FG_WRAP in c: + fg_wrap += 1 + if (m.get("tool_name") or "") == "terminal" and _DEADLINE.search(c): + tool_deadline += 1 + top_bytes = sorted(by_tool_bytes.items(), key=lambda kv: -kv[1])[:8] + report: Dict[str, Any] = {"observed": { + "tool_results": sum(by_tool_calls.values()), + "output_bytes_by_tool_top8": {k: v for k, v in top_bytes}, + "hardline_blocks": hardline, "hardline_blocks_malformed_class": malformed, + "foreground_timeout_refusals": fg_timeout, "foreground_timeout_requested_top": dict(fg_timeout_requested.most_common(5)), + "foreground_ampersand_refusals": fg_amp, "foreground_wrapper_refusals": fg_wrap, + "terminal_deadline_kills": tool_deadline, + "write_file": {"calls": write_calls, "chars": write_chars, "writes_over_20k_chars": big_writes, + "rewrites_of_a_file_read_this_session_over_20k": rewrite_of_existing, + "output_usd_at_fitted_price": round(write_chars / 3.5 * run.price_per_token["output_tokens"], 2)}, + "patch_calls": by_tool_calls.get("patch", 0), + }} + path = run.write("tools.json", report) + o = report["observed"] + print(f"[tools] {o['tool_results']:,} tool calls; hardline blocks {o['hardline_blocks']} ({o['hardline_blocks_malformed_class']} 'malformed' class); " + f"foreground timeout refusals {o['foreground_timeout_refusals']} (asked: {o['foreground_timeout_requested_top']}); '&' {o['foreground_ampersand_refusals']}, nohup {o['foreground_wrapper_refusals']}") + w = o["write_file"] + print(f"[tools] write_file {w['calls']:,} calls, {w['chars']/1e6:.1f}M chars (~${w['output_usd_at_fitted_price']:,}); >20k: {w['writes_over_20k_chars']}, of which rewrites of a file read this session: {w['rewrites_of_a_file_read_this_session_over_20k']}; patch calls {o['patch_calls']:,}") + print(f"[tools] wrote {path}") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/evals/postmortem/live_ab/__init__.py b/evals/postmortem/live_ab/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evals/postmortem/live_ab/auth_stampede.py b/evals/postmortem/live_ab/auth_stampede.py new file mode 100644 index 0000000000..1719faa023 --- /dev/null +++ b/evals/postmortem/live_ab/auth_stampede.py @@ -0,0 +1,79 @@ +"""Live A/B of the Nous hourly-expiry stampede, no real network. + +Local server: accepts bearer FRESH, returns 401 {"type":"authentication_error", "Your API key is invalid, +blocked or out of funds..."} for any other bearer. resolve_nous_runtime_credentials is patched to return +FRESH (standing in for the auth store the keepalive/peers have refreshed). N agents are built holding a +STALE JWT that expires in 30 s and each fires one API call concurrently. Count 401s the server saw. + +Usage: python stampede_ab.py +""" +import base64, json, os, sys, tempfile, threading, time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +root, n = sys.argv[1], int(sys.argv[2]) +sys.path.insert(0, root) +os.environ["HERMES_HOME"] = tempfile.mkdtemp(prefix="hh-") +os.environ["HERMES_STREAM_RETRIES"] = "0" + + +def jwt(exp, sub="acct-A"): + # `sub` matters: pre-expiry adoption (#103526 round 2) only swaps to a key for the SAME account. + b = lambda o: base64.urlsafe_b64encode(json.dumps(o).encode()).rstrip(b"=").decode() # noqa: E731 + return f"{b({'alg':'none'})}.{b({'exp':exp,'sub':sub})}.s" + + +STALE, FRESH = jwt(time.time() + 30), jwt(time.time() + 3600) +hits = {"401": 0, "200": 0} +lock = threading.Lock() + + +class H(BaseHTTPRequestHandler): + def log_message(self, *a): pass + def do_POST(self): + self.rfile.read(int(self.headers.get("content-length", 0))) + auth = self.headers.get("authorization", "") + if not self.path.endswith("/chat/completions"): + self.send_response(404); self.send_header("content-length", "0"); self.end_headers(); return + if auth == f"Bearer {FRESH}": + body = json.dumps({"id": "x", "object": "chat.completion", "created": 0, "model": "m", + "choices": [{"index": 0, "message": {"role": "assistant", "content": "ok"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}}).encode() + with lock: hits["200"] += 1 + self.send_response(200) + else: + if os.environ.get("TRACE401"): + import traceback; sys.stderr.write("401 path: "+self.path+"\n") + body = json.dumps({"error": {"type": "authentication_error", "message": "Your API key is invalid, blocked or out of funds. Please go visit the portal to sort that out: https://portal.nousresearch.com "}}).encode() + with lock: hits["401"] += 1 + self.send_response(401) + self.send_header("content-type", "application/json"); self.send_header("content-length", str(len(body))); self.end_headers(); self.wfile.write(body) + + +srv = ThreadingHTTPServer(("127.0.0.1", 0), H); threading.Thread(target=srv.serve_forever, daemon=True).start() +base = f"http://127.0.0.1:{srv.server_address[1]}/v1" + +import hermes_cli.auth as auth_mod +auth_mod.resolve_nous_runtime_credentials = lambda **kw: {"api_key": FRESH, "base_url": base} +import hermes_cli.nous_auth_keepalive as ka +ka.start_nous_auth_keepalive = lambda **kw: None # thread itself is out of scope here; we test adoption + +from run_agent import AIAgent + +agents = [] +for i in range(n): + a = AIAgent(api_key=STALE, base_url=base, provider="nous", model="test/model", quiet_mode=True, + skip_context_files=True, skip_memory=True) + a.api_mode = "chat_completions" + a._interrupt_requested = False + agents.append(a) + +results = [] +def go(a): + try: + results.append(a.chat("hi")[:20]); results.append("KEY:"+a.api_key[:5]) + except Exception as e: + results.append(f"ERR {type(e).__name__}") + +ts = [threading.Thread(target=go, args=(a,)) for a in agents] +t0 = time.time(); [t.start() for t in ts]; [t.join(120) for t in ts] +print("keys:", sorted(set(r for r in results if r.startswith("KEY"))));print(f"agents={n} server_401={hits['401']} server_200={hits['200']} errors={sum(r.startswith('ERR') for r in results)} wall={time.time()-t0:.1f}s") diff --git a/evals/postmortem/live_ab/batch_failure_notice.py b/evals/postmortem/live_ab/batch_failure_notice.py new file mode 100644 index 0000000000..d713e8f3bf --- /dev/null +++ b/evals/postmortem/live_ab/batch_failure_notice.py @@ -0,0 +1,42 @@ +"""Live: detached delegate_task batch, one child fails immediately, siblings run ~8 s. Does the parent's +completion queue see the per-task failure notice BEFORE the consolidated batch result? +Usage: python notice_live.py """ +import json, os, sys, tempfile, threading, time +root = sys.argv[1]; sys.path.insert(0, root) +os.environ["HERMES_HOME"] = tempfile.mkdtemp(prefix="hh-") +os.environ["HERMES_STREAM_RETRIES"] = "0" + +from run_agent import AIAgent +import tools.delegate_tool as dt +import tools.delegate_tool_dispatch as dd +from tools.process_registry import process_registry + +parent = AIAgent(api_key="k", base_url="https://example.com/v1", provider="test-provider", model="test/model", + quiet_mode=True, skip_context_files=True, skip_memory=True) +parent.api_mode = "chat_completions" + +# Stub the child run: task 0 fails instantly, others take 8 s and succeed. +orig_run = dd._Batch.run_child if hasattr(dd, "_Batch") else None +def fake_run_child(self, idx, task, child): + if idx == 0: + return {"task_index": 0, "status": "error", "error": "simulated 401 authentication_error", "api_calls": 0, "duration_seconds": 0.1} + time.sleep(8) + return {"task_index": idx, "status": "completed", "summary": f"ok {idx}", "api_calls": 1, "duration_seconds": 8.0} +dd._Batch.run_child = fake_run_child + +# Background dispatch requires an async-capable session; emulate a CLI session key. +import tools.async_delegation as ad +t0 = time.time() +out = dt.delegate_task(tasks=[{"goal": "fail fast on a simulated 401 error"}, {"goal": "slow successful worker number one"}, {"goal": "slow successful worker number two"}], background=True, parent_agent=parent) +print("dispatch:", json.dumps(json.loads(out))[:160]) +seen = [] +deadline = time.time() + 30 +while time.time() < deadline: + for evt, text in process_registry.drain_notifications(session_key=""): + kind = "TASK_FAILURE_NOTICE" if evt.get("task_failure_notice") else ("BATCH_FINAL" if evt.get("is_batch") else evt.get("type")) + seen.append((round(time.time() - t0, 1), kind)) + print(f" t+{time.time()-t0:5.1f}s {kind}: {text.splitlines()[0][:100]}") + if any(k == "BATCH_FINAL" for _, k in seen): + break + time.sleep(0.3) +print("ORDER:", seen) diff --git a/evals/postmortem/live_ab/cache_prefix_live.py b/evals/postmortem/live_ab/cache_prefix_live.py new file mode 100644 index 0000000000..e5e3554302 --- /dev/null +++ b/evals/postmortem/live_ab/cache_prefix_live.py @@ -0,0 +1,49 @@ +"""Live A/B for the thinking-strip cache miss (F0). Runs a short real tool loop through AIAgent on +Fable 5.1 via Nous and prints per-call cache hit ratios from agent.log. ~10 calls, well under $1. +Arm A = current code. Arm B = HERMES_KEEP_ALL_THINKING=1 monkeypatch of _manage_thinking_signatures +that passes thinking blocks back unchanged for the Nous/Anthropic route.""" +import os, sys, re, time, subprocess, json +# LIVE: real provider calls (cents). Usage: python cache_prefix_live.py +os.environ.setdefault("HERMES_HOME", os.path.expanduser("~/.hermes")) +sys.path.insert(0, sys.argv[1]) +arm = sys.argv[2] if len(sys.argv) > 2 else "A" +import agent.anthropic_message_convert as amc +if arm == "B": + _orig = amc._manage_thinking_signatures + def _keep_all(result, base_url, model): + # keep-all model on a signature-validating route: pass every thinking block back unchanged, + # only drop cache_control on thinking blocks and the internal flag. + for idx, m in amc._assistant_block_lists(result): + for b in m["content"]: + if amc._block_type(b) in amc._THINKING_TYPES: + b.pop("cache_control", None) + m.pop("_thinking_signature_invalidated", None) + amc._manage_thinking_signatures = _keep_all + # the adapter imports the name at call time via module attribute? verify binding + import agent.anthropic_adapter as ad + if hasattr(ad, "_manage_thinking_signatures"): + ad._manage_thinking_signatures = _keep_all +from run_agent import AIAgent +from hermes_cli.runtime_provider import resolve_runtime_provider +rt = resolve_runtime_provider(requested="nous", target_model="anthropic/claude-fable-5.1") +sid = f"f0ab_{arm}_{int(time.time())}" +ag = AIAgent(model="anthropic/claude-fable-5.1", provider="nous", base_url=rt.get("base_url"), api_key=rt.get("api_key"), + api_mode=rt.get("api_mode"), session_id=sid, quiet_mode=True, + enabled_toolsets=["file", "terminal"], platform="cli", max_iterations=12, + skip_context_files=True, skip_memory=True, reasoning_config={"enabled": True, "effort": "medium"}) +task = ("In /tmp/f0ab_work (create it), do these steps ONE tool call at a time, no parallel calls: " + "1) write a.txt with 'alpha', 2) write b.txt with 'beta', 3) read a.txt, 4) read b.txt, " + "5) run `ls -la /tmp/f0ab_work`, 6) run `wc -c /tmp/f0ab_work/*`, 7) run `cat /tmp/f0ab_work/a.txt`, " + "then reply with one line: DONE.") +t0 = time.time() +r = ag.run_conversation(task) +print("final:", (r.get("final_response") or "")[:80], "| wall", round(time.time() - t0, 1), "s") +time.sleep(1) +log = subprocess.run(f"grep -h '\\[{sid}\\]' ~/.hermes/logs/agent.log | grep 'API call #'", shell=True, capture_output=True).stdout.decode("utf-8", "replace") +rows = re.findall(r"API call #(\d+): .*in=(\d+) out=(\d+) .*cache=(\d+)/(\d+) \((\d+)%\)", log) +tot_in = tot_c = 0 +for n, i, o, c, ct, p in rows: + i, c = int(i), int(c); tot_in += i; tot_c += c + print(f" call {n:>2} in={i:>7} out={o:>5} cached={c:>7} ({p}%) uncached={i-c}") +print(f"ARM {arm}: calls={len(rows)} input={tot_in} cached={tot_c} uncached={tot_in-tot_c} hit={100*tot_c/max(tot_in,1):.1f}%") +print(json.dumps({"arm": arm, "calls": len(rows), "input": tot_in, "cached": tot_c})) diff --git a/evals/postmortem/live_ab/cache_prefix_wire.py b/evals/postmortem/live_ab/cache_prefix_wire.py new file mode 100644 index 0000000000..127350eab5 --- /dev/null +++ b/evals/postmortem/live_ab/cache_prefix_wire.py @@ -0,0 +1,79 @@ +"""Definitive F0 test: capture consecutive wire payloads on a real Fable 5.1 tool loop and diff the +message prefix between call N and N+1. If Hermes strips prior-turn thinking, call N+1's messages[:k] +will NOT equal call N's messages (prefix divergence) even though the conversation only grew. +Also reports cache hit per call. Cost: a handful of calls.""" +import os, sys, re, time, json, copy, subprocess +# LIVE: makes ~6 real calls to the configured provider (a few cents). Usage: +# python cache_prefix_wire.py [--hermes-home DIR] (default HERMES_HOME: the real one, for credentials) +sys.path.insert(0, sys.argv[1]) +os.environ.setdefault("HERMES_HOME", os.path.expanduser("~/.hermes")) +if "--hermes-home" in sys.argv: + os.environ["HERMES_HOME"] = sys.argv[sys.argv.index("--hermes-home") + 1] +arm = sys.argv[2] if len(sys.argv) > 2 else "A" +import agent.anthropic_message_convert as amc +if arm == "B": + def _keep_all(result, base_url, model): + for idx, m in amc._assistant_block_lists(result): + for b in m["content"]: + if amc._block_type(b) in amc._THINKING_TYPES: b.pop("cache_control", None) + m.pop("_thinking_signature_invalidated", None) + amc._manage_thinking_signatures = _keep_all +# Capture every outbound Anthropic-format payload +captured = [] +import agent.anthropic_adapter as ad +_orig_convert = None +for name in ("convert_messages_to_anthropic", "to_anthropic_messages", "convert_to_anthropic", "build_anthropic_messages"): + if hasattr(amc, name): _orig_convert = (name, getattr(amc, name)); break +if _orig_convert is None: + # find the public converter: the function that calls _manage_thinking_signatures at line ~703 + import inspect + for name, fn in inspect.getmembers(amc, inspect.isfunction): + try: + if "_manage_thinking_signatures(result" in inspect.getsource(fn): _orig_convert = (name, fn); break + except Exception: pass +name, fn = _orig_convert +def _wrapped(*a, **k): + out = fn(*a, **k) + try: + msgs = out[1] if isinstance(out, tuple) else out # (system, messages) + captured.append(copy.deepcopy(msgs)) + except Exception: pass + return out +setattr(amc, name, _wrapped) +if hasattr(ad, name): setattr(ad, name, _wrapped) +print("hooked converter:", name) +from run_agent import AIAgent +from hermes_cli.runtime_provider import resolve_runtime_provider +rt = resolve_runtime_provider(requested="nous", target_model="anthropic/claude-fable-5.1") +sid = f"f0wire_{arm}_{int(time.time())}" +ag = AIAgent(model="anthropic/claude-fable-5.1", provider="nous", base_url=rt.get("base_url"), api_key=rt.get("api_key"), + api_mode=rt.get("api_mode"), session_id=sid, quiet_mode=True, enabled_toolsets=["file", "terminal"], + platform="cli", max_iterations=10, skip_context_files=True, skip_memory=True, + reasoning_config={"enabled": True, "effort": "medium"}) +task = ("Work in /tmp/f0wire (create it). Before EACH tool call, think carefully for a moment about edge cases. " + "Steps, one tool call each: 1) write notes.md with a 5-line summary of what a Python context manager is; " + "2) read it back; 3) run `wc -l /tmp/f0wire/notes.md`; 4) append one more line to notes.md explaining __exit__ return values; " + "5) read it back; then reply DONE.") +ag.run_conversation(task) +time.sleep(1) +log = subprocess.run(f"grep -h '\\[{sid}\\]' ~/.hermes/logs/agent.log | grep 'API call #'", shell=True, capture_output=True).stdout.decode("utf-8", "replace") +rows = re.findall(r"API call #(\d+): .*in=(\d+) out=(\d+) .*cache=(\d+)/(\d+) \((\d+)%\)", log) +for r in rows: print(f" call {r[0]:>2} in={int(r[1]):>6} out={int(r[2]):>5} cached={int(r[3]):>6} ({r[5]}%) uncached={int(r[1])-int(r[3])}") +print(f"ARM {arm}: captured {len(captured)} payloads") +# prefix divergence check +def sig(m): # message signature: role + block types + text/thinking lengths + signature presence + c = m.get("content") + if isinstance(c, list): + return (m.get("role"), tuple((b.get("type"), len(str(b.get("text", b.get("thinking", b.get("input", ""))))), bool(b.get("signature"))) for b in c)) + return (m.get("role"), str(c)[:200]) +div = 0 +for i in range(1, len(captured)): + prev, cur = captured[i-1], captured[i] + k = len(prev) + same = [sig(a) == sig(b) for a, b in zip(prev, cur[:k])] + if not all(same): + j = same.index(False); div += 1 + print(f" payload {i}: prefix DIVERGED at message {j}/{k}: prev={sig(prev[j])} cur={sig(cur[j])}") +print(f"ARM {arm}: {div} of {len(captured)-1} consecutive payloads had a mutated prefix (thinking blocks in prev assistant msgs: " + f"{sum(1 for m in captured[-1] if m.get('role')=='assistant' and isinstance(m.get('content'),list) and any(b.get('type')=='thinking' for b in m['content']))} of " + f"{sum(1 for m in captured[-1] if m.get('role')=='assistant')} assistant msgs in final payload)") diff --git a/evals/postmortem/live_ab/goal_judge_wait.py b/evals/postmortem/live_ab/goal_judge_wait.py new file mode 100644 index 0000000000..bf55692390 --- /dev/null +++ b/evals/postmortem/live_ab/goal_judge_wait.py @@ -0,0 +1,19 @@ +"""Live judge A/B on the run's real "waiting" response shape. Usage: python judge_ab.py [n]""" +import os, sys +root = sys.argv[1]; n = int(sys.argv[2]) if len(sys.argv) > 2 else 3 +sys.path.insert(0, root) +os.environ.setdefault("HERMES_HOME", os.path.expanduser("~/.hermes")) # LIVE: real auxiliary judge calls (cents) +from hermes_cli import goals +goal = ("Simplify the hermes-agent codebase by >=30% LOC with zero behavior change, decomposing every god file, " + "through parallel subagent waves; integrate, run the full test suite, and open one PR.") +response = ("Round 3 status: 4 worker batches are still running (deleg_9f1a2b, deleg_77c0de, deleg_a1b2c3, deleg_0e9f88; " + "8 workers on adapters/gateway god files, median 38 min in). Round-2 integration merged clean; full suite " + "green on the integration branch. Nothing else is dispatchable until these return: their summaries decide " + "which files round 4 takes. Waiting on the batch-complete notifications; no action needed from me now.") +kw = {"active_delegations": 4} if "active_delegations" in goals.judge_goal.__code__.co_varnames else {} +verdicts = [] +for _ in range(n): + v, reason, pf, directive, tf = goals.judge_goal(goal, response, **kw) + verdicts.append((v, directive.get("seconds") if directive else None)) + print(v, directive, "|", reason[:90]) +print("SUMMARY", verdicts) diff --git a/evals/postmortem/live_ab/hardline_scanner_matrix.py b/evals/postmortem/live_ab/hardline_scanner_matrix.py new file mode 100644 index 0000000000..7e14e17dc5 --- /dev/null +++ b/evals/postmortem/live_ab/hardline_scanner_matrix.py @@ -0,0 +1,26 @@ +import sys; sys.path.insert(0, sys.argv[1] if len(sys.argv) > 1 else ".") # usage: python hardline_scanner_matrix.py +from tools.approval_detection import detect_hardline_command as d +cases = { + # the reviewer's witnesses + "newline-hidden reboot in quoted $(grep)": ('echo "$(grep -P \'safe\' /dev/null\nreboot)"', True), + "grep with backtick operand": ("grep -e `echo needle` file", False), + # the original class (must stay fixed) + "canonical sed/grep/cut": ('sed -n "$(grep -n X f | cut -d: -f1),+3p" f', False), + "grep -c inside $()": ('echo "$(grep -c x f)"', False), + # controls: other separators inside the substitution + "; hidden reboot": ('echo "$(grep x f; reboot)"', True), + "&& hidden reboot": ('echo "$(grep x f && reboot)"', True), + "| hidden shutdown": ('echo "$(grep x f | shutdown -h now)"', True), + "nested $( ) newline reboot": ('echo "$(echo $(grep x f)\nreboot)"', True), + "backtick newline reboot": ('echo "`grep x f\nreboot`"', True), + # data that must stay allowed + "reboot as quoted grep pattern": ("grep -F 'sudo reboot' notes.md", False), + "commit msg with reboot on a data line": ('git commit -m "fix\nsudo reboot handling"', False), +} +bad = 0 +for name, (cmd, want_block) in cases.items(): + got, why = d(cmd) + ok = got == want_block + bad += not ok + print(("OK " if ok else "FAIL") + f" {name}: blocked={got} ({why})") +print("ALL OK" if not bad else f"{bad} FAILURES") diff --git a/evals/postmortem/live_ab/nested_delegate_deadline.py b/evals/postmortem/live_ab/nested_delegate_deadline.py new file mode 100644 index 0000000000..f9629799ef --- /dev/null +++ b/evals/postmortem/live_ab/nested_delegate_deadline.py @@ -0,0 +1,45 @@ +"""Live A/B for the nested-delegate deadline. A depth-1 orchestrator child dispatches one leaf that runs a +~460 s task (longer than the 420 s sequential deadline). On main the orchestrator's delegate_task call returns +'timed out after 420.0s' and the leaf runs on as an orphan; on the branch the call blocks and returns the +real result. Uses glm-5.3 via Nous for cost. Deadline shortened via config to keep the run short.""" +import os, sys, json, time, re, subprocess +# Usage: python nested_delegate_deadline.py (run once per ref; LIVE: a couple of real child calls) +root = sys.argv[1]; arm = os.path.basename(os.path.normpath(root)) +sys.path.insert(0, root) +# Temp HERMES_HOME with the real auth + a config that shortens the generic sequential deadline to 40 s, so the +# run takes ~1.5 min instead of 8. The fix exempts delegate_task from this deadline entirely, so the shortened +# value is exactly what main will hit. +import shutil, tempfile, yaml +home = tempfile.mkdtemp(prefix="dl_home_"); os.environ["HERMES_HOME"] = home +real_home = os.environ.get("HERMES_HOME_SOURCE", os.path.expanduser("~/.hermes")) # credentials are copied from here into a temp home +shutil.copy(f"{real_home}/auth.json", f"{home}/auth.json") +cfg = yaml.safe_load(open(f"{real_home}/config.yaml", encoding="utf-8")) or {} +cfg.setdefault("timeouts", {}).setdefault("tools", {})["sequential_call"] = 40 +cfg.setdefault("delegation", {})["orchestrator_enabled"] = True +yaml.safe_dump(cfg, open(f"{home}/config.yaml", "w", encoding="utf-8")) +import agent.tool_executor as te +assert te.__file__.startswith(root) +from agent.deadline import resolve_timeout +print("effective sequential deadline:", resolve_timeout("tools.sequential_call", default=te._resolve_concurrent_tool_timeout())) +from run_agent import AIAgent +from hermes_cli.runtime_provider import resolve_runtime_provider +MODEL = "z-ai/glm-5.3-flash" +rt = resolve_runtime_provider(requested="nous", target_model=MODEL) +sid = f"dl_{arm}_{int(time.time())}" +ag = AIAgent(model=MODEL, provider="nous", base_url=rt.get("base_url"), api_key=rt.get("api_key"), api_mode=rt.get("api_mode"), + session_id=sid, quiet_mode=True, enabled_toolsets=["terminal", "delegation"], platform="cli", max_iterations=8, + skip_context_files=True, skip_memory=True) +# Make this agent a depth-1 orchestrator exactly as delegate_tool_child_run does for a real nested orchestrator: +# at depth>0 delegate_task runs synchronously inside the tool call, which is the path under the deadline. +ag._delegate_depth = 1 +ag._delegate_role = "orchestrator" +task = ("Use delegate_task exactly once (not background) with a single task whose goal is: " + "\"Run the shell command `sleep 75 && echo LEAF_DONE_MARKER` with the terminal tool (background=false is fine, it is under the tool timeout), " + "then reply with the exact text the command printed.\" " + "When delegate_task returns, reply with ONE line: RESULT: followed by the child's summary text verbatim (or the error text if it errored).") +t0 = time.time(); r = ag.run_conversation(task); wall = time.time() - t0 +final = (r.get("final_response") or "").strip() +print(f"ARM {arm}: wall={wall:.0f}s final={final[:300]!r}") +timed_out = "timed out" in final.lower() +got_marker = "LEAF_DONE_MARKER" in final +print(json.dumps({"arm": arm, "delegate_timed_out": timed_out, "orchestrator_got_leaf_result": got_marker, "wall_s": round(wall)})) diff --git a/evals/postmortem/live_ab/subagent_context_cap.py b/evals/postmortem/live_ab/subagent_context_cap.py new file mode 100644 index 0000000000..b18223d4ac --- /dev/null +++ b/evals/postmortem/live_ab/subagent_context_cap.py @@ -0,0 +1,32 @@ +"""Live: build a real child through delegate_tool's spawn path (real imports, temp HERMES_HOME) and read the +trigger it resolves on a 1M-window model. Run against main and the branch.""" +import os, sys, tempfile, shutil +root = sys.argv[1] +sys.path.insert(0, root) +home = tempfile.mkdtemp(prefix="hh-") +os.environ["HERMES_HOME"] = home +os.environ["HERMES_STREAM_RETRIES"] = "0" +try: + from run_agent import AIAgent + import tools.delegate_tool as dt + parent = AIAgent(api_key="k", base_url="https://example.com/v1", provider="test-provider", + model="anthropic/claude-fable-5.1", quiet_mode=True, skip_context_files=True, skip_memory=True) + # find the child-construction function by name + fn = [getattr(dt, n) for n in dir(dt) if n.startswith("_") and "child" in n.lower() and callable(getattr(dt, n)) and "spawn" in (getattr(dt, n).__doc__ or "").lower() + n.lower()] + import inspect + cands = [n for n, f in inspect.getmembers(dt, inspect.isfunction) if "AIAgent(" in inspect.getsource(f)] + print("constructor fn:", cands) + f = getattr(dt, cands[0]) + sig = inspect.signature(f); print("sig:", sig) + kwargs = {} + for name, p in sig.parameters.items(): + if name == "parent_agent": kwargs[name] = parent + elif name == "goal": kwargs[name] = "hi" + elif p.default is inspect._empty: kwargs[name] = None + child = f(**kwargs) + child = child[0] if isinstance(child, tuple) else child + cc = child.context_compressor + print(f"window={cc.context_length:,} threshold_percent={cc.threshold_percent} trigger={cc.threshold_tokens:,} cap={cc.threshold_tokens_cap}") + print(f"parent trigger={parent.context_compressor.threshold_tokens:,}") +finally: + shutil.rmtree(home, ignore_errors=True) diff --git a/evals/postmortem/review_probes/__init__.py b/evals/postmortem/review_probes/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evals/postmortem/review_probes/cache_estimator_probe.py b/evals/postmortem/review_probes/cache_estimator_probe.py new file mode 100644 index 0000000000..47ecd61881 --- /dev/null +++ b/evals/postmortem/review_probes/cache_estimator_probe.py @@ -0,0 +1,38 @@ +"""#103476: preflight estimate vs Anthropic wire size with retained thinking + +Independent-review probe (written by the /review subagent for tracking issue #103563, adapted here). +It reproduced a defect in the first version of the PR; the fixed head must pass it. Paths are taken +from the command line / environment, never hard-coded. Usage: see the argument parsing at the top of the file. +""" +import os,sys,tempfile,copy,json,types,subprocess +from pathlib import Path +sys.path.insert(0,sys.argv[1] if len(sys.argv)>1 else os.getcwd());os.environ['HERMES_HOME']=tempfile.mkdtemp(prefix='cache-boundary-') # usage: +from agent.anthropic_message_convert import convert_messages_to_anthropic +from agent.context_compressor import ContextCompressor +from agent.turn_context import _preflight_request_tokens +from agent.model_metadata import estimate_messages_tokens_rough +model='anthropic/claude-fable-5.1';url='https://inference-api.nousresearch.com/v1' +cc=ContextCompressor(model=model,provider='nous',base_url=url,api_mode='anthropic_messages',config_context_length=1000000,threshold_tokens_cap=200000) +rows=[{'role':'user','content':'Investigate repository.'}] +for i in range(48): + text='reasoning text ' * 1800 + rows.extend([{'role':'assistant','content':'tool','reasoning':text,'reasoning_details':[{'type':'thinking','thinking':text,'signature':f'signature-{i}'}],'tool_calls':[{'id':f't{i}','type':'function','function':{'name':'terminal','arguments':'{}'}}]},{'role':'tool','tool_call_id':f't{i}','content':f'Result {i}'}]) +a=types.SimpleNamespace(api_mode='anthropic_messages',provider='nous',model=model,base_url=url,tools=[],_usage_anchor=None) +pre=_preflight_request_tokens(a,rows,'');wire=convert_messages_to_anthropic(rows,base_url=url,model=model)[1];wire_est=estimate_messages_tokens_rough(wire) +print(json.dumps({'head':subprocess.check_output(['git','rev-parse','HEAD'],text=True, encoding='utf-8', errors='replace').strip(),'preflight':pre,'wire_estimate':wire_est,'preflight_should_compress':cc.should_compress(pre),'wire_estimate_should_compress':cc.should_compress(wire_est)})) +# Local signature-prefix checker validates the documented contract only; it is not the provider. +def prefixes(messages): + before=[]; result={} + for msg in messages: + nonthinking=[] + for b in msg['content'] if isinstance(msg['content'],list) else [{'type':'text','text':msg['content']}]: + if b.get('type')=='thinking':result[b['signature']]=json.dumps(before,sort_keys=True) + else:nonthinking.append(b) + before.append({'role':msg['role'],'content':nonthinking}) + return result +before=prefixes(wire) +cc._generate_summary=lambda *a,**k:'Task: investigate repository. Continue reviewing remaining files.' +out=cc.compress(copy.deepcopy(rows),current_tokens=wire_est,force=True) +after=prefixes(convert_messages_to_anthropic(out,base_url=url,model=model)[1]) +print('retained',list(after),'prefix_changed',[k for k,v in after.items() if before.get(k)!=v]) +print('retained with unchanged system; actual agent rebuilds system too, so this is lower bound') diff --git a/evals/postmortem/review_probes/context_cap_probe.py b/evals/postmortem/review_probes/context_cap_probe.py new file mode 100644 index 0000000000..57b3300f6a --- /dev/null +++ b/evals/postmortem/review_probes/context_cap_probe.py @@ -0,0 +1,95 @@ +"""#103513: child cap through repeated compression and persistence + +Independent-review probe (written by the /review subagent for tracking issue #103563, adapted here). +It reproduced a defect in the first version of the PR; the fixed head must pass it. Paths are taken +from the command line / environment, never hard-coded. Usage: see the argument parsing at the top of the file. +""" +import os, sys, tempfile, json, socket, copy +from pathlib import Path +from types import SimpleNamespace +root, tag = sys.argv[1:3] +sys.path.insert(0, root) +os.chdir(root) +for k in list(os.environ): + if any(s in k for s in ('API_KEY','TOKEN','SECRET')) or k.startswith('HERMES_'): + os.environ.pop(k, None) +home = tempfile.mkdtemp(prefix='cap-review-') +os.environ['HERMES_HOME'] = home +os.environ['HERMES_DISABLE_REDACTION'] = 'true' +import yaml +cfg = {'model': {'default': 'anthropic/claude-fable-5.1', 'provider':'openai-compat', 'base_url':'http://127.0.0.1:1/v1', 'context_length':1000000}, 'compression':{'threshold':0.85}, 'delegation': {}} +if len(sys.argv)>3: + cfg['delegation']['compression_threshold_tokens'] = json.loads(sys.argv[3]) +Path(home,'config.yaml').write_text(yaml.safe_dump(cfg), encoding='utf-8') +def blocked(*a, **kw): + raise RuntimeError('Network disabled in unpaid cap probe') +socket.socket.connect = blocked +socket.create_connection = blocked +from run_agent import AIAgent +import tools.delegate_tool as dt +import agent.context_compressor as mod +from agent.model_metadata import estimate_messages_tokens_rough +from hermes_state import SessionDB +from unittest.mock import patch +print('IDENTITY',json.dumps({'tag':tag,'tree':root,'delegate':dt.__file__,'compressor':mod.__file__,'cap_present':hasattr(dt,'_apply_child_compression_cap'),'home':home}),flush=True) +db=SessionDB(Path(home,'state.db')) +parent=AIAgent(api_key='test-key',base_url='http://127.0.0.1:1/v1',provider='openai-compat',model='anthropic/claude-fable-5.1',enabled_toolsets=[],quiet_mode=True,skip_context_files=True,skip_memory=True,save_trajectories=False,session_db=db) +child=dt._build_child_agent(task_index=0,goal='Review compression cap; continue the task.',context=None,toolsets=[],model=None,max_iterations=10,task_count=1,parent_agent=parent) +cc=child.context_compressor +print('SPAWN',json.dumps({'parent':parent.context_compressor.threshold_tokens,'child':cc.threshold_tokens,'cap':cc.threshold_tokens_cap,'tail':cc.tail_token_budget,'enabled':child.compression_enabled}),flush=True) +if len(sys.argv)>3: + child.close(); parent.close(); db.close(); sys.exit(0) +# Real trigger and real compressor transitions, no hand-written compression behavior. +base=[{'role':'user','content':'Continue reviewing this project and preserve the current goal.'}] +for i in range(48): + base.append({'role':'assistant' if i%2==0 else 'user','content':f'fixture {i} '+('A documented implementation detail with evidence and constraints. '*360)}) +base.append({'role':'user','content':'Continue the review and report findings.'}) +summary_calls=[] +def local_summary(**kw): + assert kw['task']=='compression' + summary_calls.append(len(kw['messages'][0]['content'])) + return SimpleNamespace(choices=[SimpleNamespace(message=SimpleNamespace(content='## Current task\nContinue reviewing the project. Preserve evidence and report findings.\n## Decisions\nNo changes have been applied.\n## Next steps\nInspect remaining details.',reasoning=None,reasoning_content=None),finish_reason='stop')],usage=None) +messages=base +for cycle in range(4): + tokens=estimate_messages_tokens_rough(messages) + cc.update_from_response({'prompt_tokens': tokens,'completion_tokens':0}) + should=cc.should_compress(tokens) + before=len(messages) + if should: + with patch.object(mod,'call_llm',side_effect=local_summary): + messages=cc.compress(messages,current_tokens=tokens) + after=estimate_messages_tokens_rough(messages) + print('CYCLE',json.dumps({'cycle':cycle,'before_tokens':tokens,'trigger':cc.threshold_tokens,'should':should,'before_messages':before,'after_tokens':after,'after_messages':len(messages),'compressions':cc.compression_count,'summary_calls':len(summary_calls),'blocked':cc.should_compress_info(tokens),'cap':cc.threshold_tokens_cap}),flush=True) + cc.update_from_response({'prompt_tokens':after,'completion_tokens':0}) + if cycle < 3: + messages=messages+copy.deepcopy(base[1:]) +print('SUMMARY_INPUT_CHARS',summary_calls,flush=True) +# Model switching must keep the cap, including ratio-lower small windows. +for window in [128000,1000000]: + cc.update_model(child.model,window,provider=child.provider,base_url=child.base_url,api_mode=child.api_mode) + print('MODEL_SWITCH',window,cc.threshold_tokens,cc.threshold_tokens_cap,flush=True) +if hasattr(dt,'_apply_child_compression_cap'): + for raw in ['200k','default',True,False,0,None,200000.9,float('inf')]: + temp=SimpleNamespace(context_compressor=mod.ContextCompressor(model=child.model,threshold_percent=.85,config_context_length=1000000,quiet_mode=True)) + try: + dt._apply_child_compression_cap(temp,{'compression_threshold_tokens':raw}) + print('CONFIG',repr(raw),temp.context_compressor.threshold_tokens,temp.context_compressor.threshold_tokens_cap,flush=True) + except Exception as exc: + print('CONFIG_EXCEPTION',repr(raw),type(exc).__name__,str(exc),flush=True) +# Accounting policy: the cap changes the threshold, not what route-aware pressure counts. +from agent.turn_context import _preflight_request_tokens, _agent_stale_thinking_on_wire +history=[{'role':'user','content':'review'}] +for i in range(24): + history += [{'role':'assistant','content':'observed','reasoning_content':'reasoning detail '*4000},{'role':'user','content':'continue'}] +for provider, model in [('openai-compat','test-model'),('deepseek','deepseek-chat')]: + route=SimpleNamespace(provider=provider,model=model,base_url='http://127.0.0.1:1/v1',api_mode='chat_completions',tools=[]) + pressure=_preflight_request_tokens(route,history,'') + print('ACCOUNTING',provider,_agent_stale_thinking_on_wire(route),pressure,cc.should_compress(pressure),flush=True) +# Already-cached legacy tail is an explicit compatibility edge. +if hasattr(dt,'_apply_child_compression_cap'): + legacy=SimpleNamespace(context_compressor=mod.ContextCompressor(model=child.model,threshold_percent=.85,config_context_length=1000000,tail_mode='legacy',quiet_mode=True)) + old_tail=legacy.context_compressor.tail_token_budget + dt._apply_child_compression_cap(legacy,{}) + print('LEGACY_RESOLVED',old_tail,legacy.context_compressor.tail_token_budget,legacy.context_compressor.threshold_tokens,flush=True) +child.close();parent.close();db.close() +print('DONE',tag,flush=True) diff --git a/evals/postmortem/review_probes/credential_identity_probe.py b/evals/postmortem/review_probes/credential_identity_probe.py new file mode 100644 index 0000000000..4dd63ddfc9 --- /dev/null +++ b/evals/postmortem/review_probes/credential_identity_probe.py @@ -0,0 +1,113 @@ +"""#103526: explicit account-A key must not be replaced by the singleton's account-B key (real SDK -> loopback) + +Independent-review probe (written by the /review subagent for tracking issue #103563, adapted here). +It reproduced a defect in the first version of the PR; the fixed head must pass it. Paths are taken +from the command line / environment, never hard-coded. Usage: see the argument parsing at the top of the file. +""" +import os, tempfile, sys, json, time, base64, threading, importlib.util +from pathlib import Path +from datetime import datetime, timezone +from http.server import ThreadingHTTPServer, BaseHTTPRequestHandler +ROOT = Path(sys.argv[1]); MODE = sys.argv[2] +sys.path.insert(0, str(ROOT)) +home = Path(tempfile.mkdtemp(prefix='pr103526-probe-')) +os.environ.update(HOME=str(home), HERMES_HOME=str(home/'hermes'), XDG_CONFIG_HOME=str(home/'config'), CODEX_HOME=str(home/'codex')) +for k in list(os.environ): + if any(x in k for x in ('API_KEY','TOKEN','NOUS_','SECRET')): os.environ.pop(k, None) +(home/'hermes').mkdir() +(home/'hermes'/'config.yaml').write_text('nous:\n keepalive_interval_seconds: 0\nmemory:\n memory_enabled: false\n', encoding='utf-8') +def guard(event,args): + if event == 'socket.connect' and isinstance(args[1], tuple) and args[1][0] not in ('127.0.0.1','::1'): + raise RuntimeError('External network forbidden by review probe') +sys.addaudithook(guard) +def jwt(sub, ttl): + def part(v): return base64.urlsafe_b64encode(json.dumps(v).encode()).decode().rstrip('=') + return part({'alg':'none'})+'.'+part({'sub':sub,'scope':'inference:invoke','exp':int(time.time()+ttl)})+'.sig' +def claims(token): + return json.loads(base64.urlsafe_b64decode(token.split('.')[1]+'===')) +records=[] +class Handler(BaseHTTPRequestHandler): + def do_POST(self): + raw_body=self.rfile.read(int(self.headers.get('Content-Length',0))) + if self.path == '/api/oauth/token': + records.append({'path':self.path,'refresh':True}) + raw=json.dumps({'access_token':refresh_reply,'refresh_token':'fixture-rotated','expires_in':3600,'token_type':'Bearer','scope':'inference:invoke'}).encode() + self.send_response(200);self.send_header('Content-Type','application/json');self.send_header('Content-Length',str(len(raw)));self.end_headers();self.wfile.write(raw);return + body=json.loads(raw_body or '{}') + bearer=self.headers.get('Authorization','').removeprefix('Bearer ') + records.append({'path':self.path,'sub':claims(bearer).get('sub') if bearer else None}) + if bearer and claims(bearer)['exp'] < time.time(): + raw=json.dumps({'error':{'message':'expired bearer','type':'authentication_error'}}).encode();self.send_response(401);self.send_header('Content-Type','application/json');self.send_header('Content-Length',str(len(raw)));self.end_headers();self.wfile.write(raw);return + data={'id':'local-probe','object':'chat.completion','created':int(time.time()),'model':'hermes-test','choices':[{'index':0,'message':{'role':'assistant','content':'local-only'},'finish_reason':'stop'}],'usage':{'prompt_tokens':1,'completion_tokens':1,'total_tokens':2}} + raw=json.dumps(data).encode(); self.send_response(200);self.send_header('Content-Type','application/json');self.send_header('Content-Length',str(len(raw)));self.end_headers();self.wfile.write(raw) + def log_message(self,*a): pass +server=ThreadingHTTPServer(('127.0.0.1',0),Handler); threading.Thread(target=server.serve_forever,daemon=True).start() +url=f'http://127.0.0.1:{server.server_port}/v1' +# Runtime override preserves loopback routing, without relaxing URL validation. +os.environ['NOUS_INFERENCE_BASE_URL']=url +os.environ['HERMES_SHARED_AUTH_DIR']=str(home/'shared') +from run_agent import AIAgent +from agent.turn_iteration_prep import prepare_iteration +import agent.client_lifecycle as lifecycle +import hermes_cli.auth as auth +print(json.dumps({'mode':MODE,'module':lifecycle.__file__,'has_new':hasattr(AIAgent,'_adopt_nous_key_before_expiry'),'home':str(home)}),flush=True) +if MODE=='main': + spec=importlib.util.spec_from_file_location('main_prep',Path(__file__).with_name('main-turn_iteration_prep.py')); mod=importlib.util.module_from_spec(spec);sys.modules[spec.name]=mod;spec.loader.exec_module(mod);prepare_iteration=mod.prepare_iteration + +def store(token): + exp=claims(token)['exp']; state={'portal_base_url':'https://portal.nousresearch.com','inference_base_url':'https://inference-api.nousresearch.com/v1','client_id':'hermes-cli','token_type':'Bearer','scope':'inference:invoke','access_token':token,'refresh_token':'fixture-refresh-never-send','expires_at':datetime.fromtimestamp(exp,timezone.utc).isoformat(),'expires_in':3600,'agent_key':token,'agent_key_expires_at':datetime.fromtimestamp(exp,timezone.utc).isoformat()} + (home/'hermes'/'auth.json').write_text(json.dumps({'version':1,'active_provider':'nous','providers':{'nous':state}}), encoding='utf-8') +results=[] +for case, own_sub, store_sub, ttl in [('same-account','account-A','account-A',30),('explicit-account','account-A','account-B',30),('far-from-expiry','account-A','account-B',3000)]: + own=jwt(own_sub,ttl);fresh=jwt(store_sub,3600);store(fresh) + agent=AIAgent(api_key=own,base_url=url,provider='nous',model='hermes-test',quiet_mode=True,skip_context_files=True,skip_memory=True,enabled_toolsets=[]) + messages=[{'role':'user','content':'local fixture'}]; before=json.dumps(messages) + prepare_iteration(agent,messages=messages,api_call_count=0) + client=agent._create_request_openai_client(reason='review_probe') + reply=client.chat.completions.create(model='hermes-test',messages=messages) + result={'case':case,'before_sub':own_sub,'after_sub':claims(agent.api_key)['sub'],'adopted_fresh':agent.api_key==fresh,'messages_unchanged':json.dumps(messages)==before,'wire':records[-1],'reply':reply.choices[0].message.content} + results.append(result); print(json.dumps(result),flush=True) + agent._close_request_openai_client(client,reason='probe_done') + agent.client.close() +# Contended peer-adoption: real auth-store locking and SDK wire, no resolver mocks. +from concurrent.futures import ThreadPoolExecutor +fresh=jwt('account-A',3600); store(fresh) +expired=jwt('account-A',-30) +barrier=threading.Barrier(12) +def worker(_): + a=AIAgent(api_key=expired,base_url=url,provider='nous',model='hermes-test',quiet_mode=True,skip_context_files=True,skip_memory=True,enabled_toolsets=[]) + barrier.wait(timeout=30) + messages=[{'role':'user','content':'concurrent local fixture'}] + prepare_iteration(a,messages=messages,api_call_count=0) + c=a._create_request_openai_client(reason='concurrent_review_probe') + try: + c.chat.completions.create(model='hermes-test',messages=messages) + status=200 + except Exception as e: + status=getattr(e,'status_code',type(e).__name__) + finally: + a._close_request_openai_client(c,reason='probe_done'); a.client.close() + return status +with ThreadPoolExecutor(max_workers=12) as executor: + statuses=list(executor.map(worker,range(12))) +print(json.dumps({'concurrent_statuses':statuses,'successes':statuses.count(200),'401s':statuses.count(401)}),flush=True) +# No fresh peer key: twelve agents contend for one real local refresh POST. +refresh_reply=jwt('account-A',3600) +if MODE != 'main': + import shutil + shutil.rmtree(home/'shared',ignore_errors=True) + expired=jwt('account-A',30); store(expired) + state_file=home/'hermes'/'auth.json' + state=json.loads(state_file.read_text(encoding='utf-8')); state['providers']['nous']['portal_base_url']=url.removesuffix('/v1');state_file.write_text(json.dumps(state), encoding='utf-8') + barrier=threading.Barrier(12);records.clear() + with ThreadPoolExecutor(max_workers=12) as executor: + mint_statuses=list(executor.map(worker,range(12))) + posts=sum(r.get('refresh',False) for r in records) + print(json.dumps({'mint_statuses':mint_statuses,'refresh_posts':posts}),flush=True) + assert posts == 1 and mint_statuses == [200]*12 +server.shutdown() +(Path(os.environ.get('PROBE_OUT', tempfile.gettempdir()))/('cred-probe-'+MODE+'.json')).write_text(json.dumps({'cases':results,'concurrent_statuses':statuses},indent=2), encoding='utf-8') +assert all(r['messages_unchanged'] for r in results) +# Regression contract (fixed head): the explicit account-A key must stay A; same-account adoption must still happen. +assert results[1]['after_sub']=='account-A', 'explicit-account key was replaced by the singleton account (the round-1 defect)' +assert results[0]['after_sub']=='account-A' and results[0]['adopted_fresh'], 'same-account fresh key must still be adopted' diff --git a/evals/postmortem/review_probes/deadline_probe.py b/evals/postmortem/review_probes/deadline_probe.py new file mode 100644 index 0000000000..630795b7ac --- /dev/null +++ b/evals/postmortem/review_probes/deadline_probe.py @@ -0,0 +1,140 @@ +"""#103486: nested delegate deadline through actual dispatch + +Independent-review probe (written by the /review subagent for tracking issue #103563, adapted here). +It reproduced a defect in the first version of the PR; the fixed head must pass it. Paths are taken +from the command line / environment, never hard-coded. Usage: see the argument parsing at the top of the file. +""" +import os, sys, tempfile, pathlib, json, time, threading, socket, subprocess +from types import SimpleNamespace as NS +root = pathlib.Path(sys.argv[1]).resolve() +sys.path.insert(0, str(root)); os.chdir(root) +for k in list(os.environ): + if k.startswith('HERMES_') or k.endswith(('_API_KEY', '_TOKEN')): + os.environ.pop(k, None) +home = tempfile.TemporaryDirectory(prefix='deadline-probe-') +os.environ['HERMES_HOME'] = home.name +os.environ['HERMES_DISABLE_TELEMETRY'] = '1' +pathlib.Path(home.name, 'config.yaml').write_text('timeouts:\n tools:\n sequential_call: 0.3\ndelegation:\n max_summary_chars: 24000\n', encoding='utf-8') +# Any accidental provider, metadata, or telemetry request fails closed. +def no_connect(*args, **kwargs): + raise RuntimeError('OFFLINE PROBE: network forbidden') +socket.socket.connect = no_connect +socket.create_connection = no_connect +import agent.tool_executor as te +import agent.turn_usage as tu +import tools.delegate_tool_results as dr +import agent.usage_pricing as up +import agent.codex_runtime as cr +from run_agent import AIAgent +for mod in (te, tu, dr, up, cr): + assert pathlib.Path(mod.__file__).is_relative_to(root), mod.__file__ +print(json.dumps({'sha': subprocess.check_output(['git','rev-parse','HEAD'],text=True, encoding='utf-8', errors='replace').strip(), 'modules': {m.__name__:m.__file__ for m in (te,tu,dr,up,cr)}, 'home':home.name}), flush=True) +a = AIAgent(api_key='offline-fixture', base_url='http://127.0.0.1:9/v1', provider='openai-compat', model='offline-test', enabled_toolsets=[], quiet_mode=True, skip_context_files=True, skip_memory=True, save_trajectories=False) +a._delegate_depth = 1; a._delegate_role = 'orchestrator' +assert te._resolve_sequential_tool_timeout() == 0.3 +rows=[] +for name in ('delegate_task','terminal','execute_code'): + entered=threading.Event(); finished=threading.Event() + def work(args): + entered.set() + time.sleep(0.65) + finished.set() + return json.dumps({'marker':'LEAF_DONE_MARKER'}) + start=time.monotonic() + result=te._run_sequential_tool_execution_middleware(a,function_name=name,function_args={},effective_task_id='offline',tool_call_id='probe-'+name,execute=work) + elapsed=time.monotonic()-start + assert entered.is_set(), 'tool body never entered' + row={'case':name,'elapsed':round(elapsed,4),'result_type':type(result.result).__name__,'result':str(result.result),'completed_on_return':finished.is_set()} + assert finished.wait(2), 'fixture did not finish' + rows.append(row) +print(json.dumps({'tool_boundary':rows}),flush=True) +# Warm real middleware, then interrupt a non-cooperative delegated operation. +a._interrupt_requested=False +entered=threading.Event(); release=threading.Event() +def blocked(args): + entered.set(); release.wait(8); return 'late' +def interrupt(): + assert entered.wait(2) + time.sleep(0.1) + a.interrupt('offline cancellation probe') +t = threading.Thread(target=interrupt); t.start() +start=time.monotonic() +r=te._run_sequential_tool_execution_middleware(a,function_name='delegate_task',function_args={},effective_task_id='offline',tool_call_id='probe-interrupt',execute=blocked) +elapsed=time.monotonic()-start +release.set();t.join() +print(json.dumps({'interrupt':{'elapsed':round(elapsed,4),'result_type':type(r.result).__name__,'result':str(r.result)}}),flush=True) +# Feed provider-shaped data to the production writer, not fabricated _last_turn_usage. +# Price lookup is unrelated and potentially networked; only pricing is stubbed. +tu.estimate_usage_cost=lambda *args,**kwargs: NS(amount_usd=None,status='unknown',source='offline') +up.estimate_usage_cost=tu.estimate_usage_cost +class Compressor: + context_length=200000 + max_tokens=8000 + threshold_tokens=190000 + def update_from_response(self, usage): + self.last_prompt_tokens=usage.get('prompt_tokens',-1) +def parent(provider,mode='chat_completions',client=None): + b=NS(context_compressor=Compressor(),provider=provider,api_mode=mode,model='offline',base_url='',client=client,session_id='offline-usage',_session_db=None,quiet_mode=True,verbose_logging=False) + for key in ('api_calls','prompt_tokens','completion_tokens','total_tokens','input_tokens','output_tokens','cache_read_tokens','cache_write_tokens','reasoning_tokens','estimated_cost_usd'): + setattr(b,'session_'+key,0) + return b +def record(b,raw): + tu.record_response_usage(b,NS(usage=raw),messages=[{'role':'user','content':'fixture'}],api_call_count=1,api_duration=0,compression_attempts=0,max_compression_attempts=3) + return {'prompt':b._last_turn_usage['prompt_tokens'],'input':b._last_turn_usage['input_tokens'],'cache_read':b._last_turn_usage['cache_read_tokens'],'cache_write':b._last_turn_usage['cache_write_tokens'],'budget':dr._parent_summary_char_budget(b,1)} +chat={'prompt_tokens':30000,'completion_tokens':100,'prompt_tokens_details':{'cached_tokens':20000,'cache_write_tokens':5000}} +anth={'input_tokens':5000,'output_tokens':100,'cache_read_input_tokens':20000,'cache_creation_input_tokens':5000} +responses={'input_tokens':30000,'output_tokens':100,'input_tokens_details':{'cached_tokens':20000,'cache_write_tokens':5000}} +cases=[('openai', 'chat_completions',chat),('nous','chat_completions',chat),('openrouter','chat_completions',chat),('anthropic','anthropic_messages',anth),('minimax','anthropic_messages',anth),('minimax-cn','anthropic_messages',anth),('bedrock','chat_completions',{'prompt_tokens':30000,'completion_tokens':100,'cache_read_input_tokens':20000,'cache_creation_input_tokens':5000}),('google','chat_completions',chat),('deepseek','chat_completions',{'prompt_tokens':30000,'completion_tokens':100,'prompt_cache_hit_tokens':20000}),('moonshot','chat_completions',{'prompt_tokens':30000,'completion_tokens':100,'cached_tokens':20000}),('openai-codex','codex_responses',responses),('openai-compat','chat_completions',{'input_tokens':30000,'output_tokens':100})] +usage_rows=[] +for provider,mode,raw in cases: + b=parent(provider,mode) + fresh=record(b,raw) + assert fresh['prompt']==30000 + b.session_prompt_tokens=25000000 + long_budget=dr._parent_summary_char_budget(b,1) + usage_rows.append({'provider':provider,'mode':mode,**fresh,'long_lived_budget':long_budget}) +# Raw native adapter response -> production conversion -> writer -> budget. +from agent.bedrock_adapter import normalize_converse_response +from agent.gemini_native_adapter import _usage_from_metadata +native=[] +for provider, raw in [('bedrock',normalize_converse_response({'usage':{'inputTokens':5000,'cacheReadInputTokens':20000,'cacheWriteInputTokens':5000,'outputTokens':100}}).usage),('google',_usage_from_metadata({'promptTokenCount':30000,'cachedContentTokenCount':25000,'candidatesTokenCount':100,'totalTokenCount':30100}))]: + b=parent(provider); values=record(b,raw); assert values['prompt']==30000 + native.append({'provider':provider,**values}) +print(json.dumps({'provider_writers':usage_rows,'native_adapters':native}),flush=True) +# Exercise the actual MoA accounting deposit/consume path without model calls. +from agent.moa_loop import MoAClient +moa_client = MoAClient('offline') +moa_client.chat.completions._fold_pending_accounting(up.CanonicalUsage(input_tokens=240000), None) +b=parent('moa',client=moa_client) +moa=record(b,chat) +results=[{'task_index':0,'summary':'X'*5000}] +dr._apply_summary_budget(results,b) +print(json.dumps({'moa':{**moa,'actual_aggregator_prompt':30000,'anchor_prompt':b._usage_anchor.prompt_tokens if hasattr(b._usage_anchor,'prompt_tokens') else repr(b._usage_anchor),'truncated':results[0].get('summary_truncated',False)}}),flush=True) +b=parent('openai-codex','codex_app_server') +b._last_turn_usage=None +codex=cr._record_codex_app_server_usage(b,NS(token_usage_last={'inputTokens':170000,'cachedInputTokens':20000,'outputTokens':100},model_context_window=200000)) +print(json.dumps({'codex_app_server':{'returned_prompt':codex['prompt_tokens'],'last_turn_usage':b._last_turn_usage,'compressor_prompt':b.context_compressor.last_prompt_tokens,'budget':dr._parent_summary_char_budget(b,1)}}),flush=True) +# Missing usage and full-context defaults, including a current turn with no usage. +b=parent('openai-compat'); record(b,{'prompt_tokens':190000,'completion_tokens':100}) +print(json.dumps({'near_full_budget':dr._parent_summary_char_budget(b,1),'near_full_batch_budget':dr._parent_summary_char_budget(b,5)}),flush=True) +# A successful provider response without usage keeps no current usage after turn reset. +b._last_turn_usage=None +record_outcome=tu.record_response_usage(b,NS(usage=None),messages=[],api_call_count=1,api_duration=0,compression_attempts=0,max_compression_attempts=3) +missing=[{'task_index':0,'summary':'X'*20000}] +dr._apply_summary_budget(missing,b) +print(json.dumps({'missing_usage_next_turn':{'session_prompt':b.session_prompt_tokens,'compressor_prompt':b.context_compressor.last_prompt_tokens,'last_turn_usage':b._last_turn_usage,'budget':dr._parent_summary_char_budget(b,1),'summary_length':len(missing[0]['summary']),'truncated':missing[0].get('summary_truncated',False)}}),flush=True) +# Full sequential executor -> resolver -> real middleware -> result commit. +# Replace only the model/delegation body; no subagents or model requests. +a._interrupt_requested=False +full_entered=threading.Event() +def full_body(args): + full_entered.set(); time.sleep(0.65) + return json.dumps({'marker':'FULL_BOUNDARY_DONE'}) +a._dispatch_delegate_task=full_body +calls=NS(tool_calls=[NS(id='full-boundary',type='function',function=NS(name='delegate_task',arguments='{}'))]) +messages=[] +start=time.monotonic() +te.execute_tool_calls_sequential(a,calls,messages,'offline-full') +assert full_entered.is_set() +print(json.dumps({'full_sequential_boundary':{'elapsed':round(time.monotonic()-start,4),'messages':messages}}),flush=True) +print('PROBE_COMPLETE',flush=True) diff --git a/evals/postmortem/review_probes/finalizer_schedule_probe.py b/evals/postmortem/review_probes/finalizer_schedule_probe.py new file mode 100644 index 0000000000..596309560f --- /dev/null +++ b/evals/postmortem/review_probes/finalizer_schedule_probe.py @@ -0,0 +1,56 @@ +"""#103507: pytest plugin forcing the consumer-first schedule (-p finalizer_schedule_probe --finalizer-probe=consumer-first) + +Independent-review probe (written by the /review subagent for tracking issue #103563, adapted here). +It reproduced a defect in the first version of the PR; the fixed head must pass it. Paths are taken +from the command line / environment, never hard-coded. Usage: see the argument parsing at the top of the file. +""" +import asyncio +import json +import pytest + + +def pytest_addoption(parser): + parser.addoption('--finalizer-probe', default='off') + + +@pytest.fixture(autouse=True) +def finalizer_schedule_probe(request, monkeypatch): + mode = request.config.getoption('--finalizer-probe') + if mode == 'off': + yield + return + from agent import relay_llm, chat_completion_helpers + print('PROBE_IMPORT', relay_llm.__file__, chat_completion_helpers.__file__) + original_provider = relay_llm.ManagedLlmStream._provider_stream + original_count = chat_completion_helpers._StreamingCall._count_chunk + stats = {'gated_terminal_chunks': 0, 'consumer_releases': 0} + + def terminal(chunk): + choices = chunk.get('choices') or [] + return (not choices and chunk.get('usage') is not None) or ( + bool(choices) and choices[0].get('finish_reason') == 'tool_calls') + + async def gated_provider(stream, *args): + async for chunk in original_provider(stream, *args): + if terminal(chunk): + stream._review_gate = asyncio.Event() + stats['gated_terminal_chunks'] += 1 + yield chunk + await asyncio.wait_for(stream._review_gate.wait(), timeout=10) + else: + yield chunk + + def count_and_release(call, diag, chunk): + stream = call.managed_stream_holder.get('stream') + gate = getattr(stream, '_review_gate', None) + raw = chunk.model_dump() if hasattr(chunk, 'model_dump') else vars(chunk) + if terminal(raw) and gate is not None and not gate.is_set(): + gate.set() + stats['consumer_releases'] += 1 + return original_count(call, diag, chunk) + + assert mode == 'consumer-first' + monkeypatch.setattr(relay_llm.ManagedLlmStream, '_provider_stream', gated_provider) + monkeypatch.setattr(chat_completion_helpers._StreamingCall, '_count_chunk', count_and_release) + yield + print('PROBE_STATS', json.dumps(stats, sort_keys=True)) diff --git a/evals/postmortem/review_probes/goal_repaste_probe.py b/evals/postmortem/review_probes/goal_repaste_probe.py new file mode 100644 index 0000000000..6f6c8c852a --- /dev/null +++ b/evals/postmortem/review_probes/goal_repaste_probe.py @@ -0,0 +1,129 @@ +"""#103553: /goal re-paste vs option-selecting fragment + +Independent-review probe (written by the /review subagent for tracking issue #103563, adapted here). +It reproduced a defect in the first version of the PR; the fixed head must pass it. Paths are taken +from the command line / environment, never hard-coded. Usage: see the argument parsing at the top of the file. +""" +import asyncio +import contextlib +import copy +import io +import json +import os +from pathlib import Path +import queue +import socket +import sqlite3 +import sys +import tempfile +import threading +from unittest.mock import patch + +repo, tag = sys.argv[1:3] +sys.path.insert(0, repo) +# Remove all inherited Hermes/config and credential env before real imports. +for key in list(os.environ): + if key.startswith('HERMES_') or key.endswith(('_API_KEY', '_TOKEN')): + os.environ.pop(key, None) +home = tempfile.TemporaryDirectory(prefix='review-goaldup-') +os.environ['HERMES_HOME'] = home.name +os.environ['NO_PROXY'] = '*' +os.environ['TZ'] = 'UTC' +# Fail closed: these probes must never invoke a provider or external network. +socket.socket.connect = lambda *a, **k: (_ for _ in ()).throw(RuntimeError('network prohibited in review probe')) +from cli import HermesCLI +from hermes_cli import cli_commands_mixin, goals +from gateway.slash_commands_goals import GatewayGoalCommandsMixin +from gateway.config import Platform +from gateway.platforms.base import MessageEvent, MessageType +from gateway.session import SessionSource +from tui_gateway import server + +assert str(Path(cli_commands_mixin.__file__).resolve()).startswith(repo) +assert str(Path(goals.__file__).resolve()).startswith(repo) +server._hermes_home = Path(home.name) +goals._DB_CACHE.clear() +goals._get_session_db() +source = sqlite3.connect('file:/tmp/rf/state_copy.db?mode=ro', uri=True) +source.row_factory = sqlite3.Row +original = source.execute('SELECT content FROM messages WHERE id=264820').fetchone()['content'] +repeated = source.execute('SELECT content FROM messages WHERE id=267045').fetchone()['content'] +assert original == repeated +history = [{'role': 'user', 'content': original}, {'role': 'assistant', 'content': 'The wave is running; I will pick this up when the workers report back.'}] +output = {'tag': tag, 'modules': [cli_commands_mixin.__file__, goals.__file__, server.__file__], 'source_equal': original == repeated, 'original_chars': len(original)} + +def make_cli(hist, sid): + c = HermesCLI.__new__(HermesCLI) + c.session_id = sid + c.agent = None + c.conversation_history = copy.deepcopy(hist) + c._pending_input = queue.Queue() + return c + +def cli_case(hist, goal, sid): + c = make_cli(hist, sid) + before = copy.deepcopy(c.conversation_history) + with contextlib.redirect_stdout(io.StringIO()): + assert c.process_command('/goal ' + goal) + prompt = c._pending_input.get_nowait() + state = goals.GoalManager(sid).state + assert c.conversation_history == before + assert c._pending_input.empty() + assert state.goal == goal + return {'prompt': prompt, 'prompt_chars': len(prompt), 'state_goal_preserved': state.goal == goal, + 'history_unchanged': c.conversation_history == before, + 'continuation_still_contains_full_goal': goal in goals.GoalManager(sid).next_continuation_prompt()} + +output['cli_literal_witness'] = cli_case(history, original, tag+'-cli') +output['cli_empty_control'] = cli_case([], 'Ship the release', tag+'-empty') +output['cli_unrelated_control'] = cli_case([{'role':'user','content':'Unrelated'}], original, tag+'-unrelated') +output['cli_block_content'] = cli_case([{'role':'user','content':[{'type':'text','text': original}]}], original, tag+'-block') +options = [{'role':'user','content':'We can ship the API or ship the UI. Wait for my choice.'}, {'role':'assistant','content':'Which one should I work on?'}] +output['selection_api'] = cli_case(options, 'ship the API', tag+'-api') +output['selection_ui'] = cli_case(options, 'ship the UI', tag+'-ui') +output['different_selected_goals_same_model_prompt'] = output['selection_api']['prompt'] == output['selection_ui']['prompt'] +# /goal draft invokes its only paid dependency as an explicit unavailable stub. +c = make_cli(history, tag+'-draft') +with patch('hermes_cli.goals.draft_contract', return_value=None), contextlib.redirect_stdout(io.StringIO()): + assert c.process_command('/goal draft ' + original) +output['draft_fallback_prompt'] = c._pending_input.get_nowait() + +# Actual registered TUI/Desktop command dispatcher; goal bypasses CLI slash worker. +sid = tag+'-tui' +server._sessions[sid] = {'session_key': sid, 'history': copy.deepcopy(history), 'history_lock': threading.Lock(), 'history_version':0, 'running':False, 'attached_images': [], 'cols': 120} +rpc = server._methods['slash.exec'](1, {'command':'goal '+original, 'session_id':sid}) +output['tui_slash_rpc'] = rpc +assert goals.GoalManager(sid).state.goal == original + +# Real gateway handler + real enqueue method. Capture only adapter transport FIFO seam. +class GatewayProbe(GatewayGoalCommandsMixin): + def __init__(self): + self.mgr = goals.GoalManager(tag+'-gateway') + self.events = [] + self.conversation_history = copy.deepcopy(history) + async def _get_goal_manager_for_event(self, event): + return self.mgr, None + def _adapter_and_key_for(self, event): + return object(), 'probe-key' + def _enqueue_fifo(self, key, event, adapter): + self.events.append(event) +gw = GatewayProbe() +event = MessageEvent(text='/goal '+original, message_type=MessageType.TEXT, + source=SessionSource(platform=Platform.DISCORD, chat_id='probe', chat_type='dm', user_id='probe'), message_id='goal-probe') +asyncio.run(gw._handle_goal_command(event)) +assert len(gw.events) == 1 +output['gateway_kick'] = {'prompt':gw.events[0].text, 'goal_preserved':gw.mgr.state.goal == original} +# Actual archived replay turn, including useful actions and terminal waits. +next_id = source.execute("SELECT min(id) FROM messages WHERE session_id=? AND id>? AND role='user'", ('20260902_073639_918cf3',267045)).fetchone()[0] +rows = source.execute('SELECT id,role,content,tool_calls,tool_name,timestamp FROM messages WHERE session_id=? AND id>=? AND id1 else os.getcwd()) # repo root under test +os.environ['HERMES_HOME']=tempfile.mkdtemp(dir=root) +os.environ['TERMINAL_ENV']='local' +from tools import file_tools as f +from tools.file_operations import WriteResult +print('module',f.__file__) +with tempfile.TemporaryDirectory(dir=root) as d: + p=pathlib.Path(d)/'file.txt' + for n in [1000,4000,10000,20000]: + old='same repetitive record\n'*n;p.write_text(old, encoding='utf-8');new=old+'end\n' + start=time.monotonic(); hint=f._whole_file_rewrite_hint('default',str(p),new) + print('repeat',n,len(old),round(time.monotonic()-start,3),bool(hint),flush=True) + old=''.join(f'{i} arbitrary unique data for test\n' for i in range(1500));p.write_text(old, encoding='utf-8') + new=old.replace('750 arbitrary','750 edited') + start=time.monotonic();r=json.loads(f.write_file_tool(str(p),new,task_id='review')) + print('live-write',round(time.monotonic()-start,3),r,p.read_text(encoding='utf-8')==new,flush=True) + class RemoteOps: + env=object() + def write_file(self,path,content): + self.written=(path,content) + return WriteResult(bytes_written=len(content),verified=True) + ops=RemoteOps() + p.write_text(old, encoding='utf-8') + with patch.object(f,'_get_file_ops',return_value=ops): + print('remote_backend_is_host',f._file_ops_uses_host_paths(ops)) + r=json.loads(f.write_file_tool(str(p),new,task_id='remote-review')) + print('remote_empty_target_hint_from_host',r.get('hint'), 'host_unchanged',p.read_text(encoding='utf-8')==old,flush=True) + fifo=pathlib.Path(d)/'pipe.txt';os.mkfifo(fifo) + alarms=[] + def alarm(sig,frame): + alarms.append(time.monotonic()) + raise TimeoutError('host FIFO read blocked') + signal.signal(signal.SIGALRM,alarm) # windows-footgun: ok — POSIX-only FIFO hazard probe + with patch.object(f,'_get_file_ops',return_value=ops): + start=time.monotonic();signal.alarm(2) + try: print('remote_fifo',f.write_file_tool(str(fifo),'x'*20000,task_id='remote-fifo'), 'seconds',time.monotonic()-start,'read_alarm_fired',bool(alarms)) + finally:signal.alarm(0) + with patch.object(f,'_whole_file_rewrite_hint',return_value=None): + start=time.monotonic();print('remote_fifo_base_no_hint',f.write_file_tool(str(fifo),'x'*20000,task_id='remote-fifo'), 'seconds',time.monotonic()-start) diff --git a/evals/postmortem/review_probes/scanner_bypass_probe.py b/evals/postmortem/review_probes/scanner_bypass_probe.py new file mode 100644 index 0000000000..aaa47a1041 --- /dev/null +++ b/evals/postmortem/review_probes/scanner_bypass_probe.py @@ -0,0 +1,55 @@ +"""#103492: a hardline command hidden after a NEWLINE inside a double-quoted $(grep …) must stay blocked +by the public guard with no approval callback; a grep with a backtick operand must stay allowed. + +Independent-review probe (written by the /review subagent for tracking issue #103563, adapted). It +reproduced an approval bypass in the first version of the PR (approved, 0 callbacks). Runs the real +public guard (check_dangerous_command) in a temp HERMES_HOME with approvals.mode=manual and executes a +HARMLESS Bash witness (reboot shadowed by a function writing a marker) to prove reachability. + +Usage: python scanner_bypass_probe.py [] +""" +import importlib.util +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +ROOT=Path(sys.argv[1]).resolve() # repo root under test +sys.path.insert(0,str(ROOT)) +home=Path(tempfile.mkdtemp(prefix='review103492-confirm-')) +os.environ['HERMES_HOME']=str(home) +os.environ['HERMES_INTERACTIVE']='1' +(home/'config.yaml').write_text('approvals:\n mode: manual\n timeout: 1\n', encoding='utf-8') +from tools import approval_detection as d +from tools import approval as a +b=None +if len(sys.argv)>2: # optional: path to a baseline approval_detection.py (e.g. from main) to compare verdicts + spec=importlib.util.spec_from_file_location('baseline',sys.argv[2]); b=importlib.util.module_from_spec(spec); spec.loader.exec_module(b) +r='re'+'boot' +cmd='echo "$(grep -P \'safe\' /dev/null\n'+r+')"' +print(json.dumps({'module':d.__file__,'approval_module':a.__file__,'case':'newline command','command':cmd,'baseline':(b.detect_hardline_command(cmd) if b else None),'head':d.detect_hardline_command(cmd),'variants':list(d._command_detection_variants(cmd))})) +calls=[] +def deny(*args,**kwargs): + calls.append([args,kwargs]);return 'deny' +print(json.dumps({'pattern_guard':a.check_dangerous_command(cmd,'local',approval_callback=deny),'combined_guard':a.check_all_command_guards(cmd,'local',approval_callback=deny),'callback_calls':calls},default=str)) +# Shadow the dangerous command name with a harmless function; verify shell +# execution and marker contents, never invoke a host shutdown binary. +marker=home/'marker' +prefix=r+'(){ printf executed > "$MARKER"; }; ' +p=subprocess.run(['/bin/bash','--noprofile','--norc','-c',prefix+cmd],env={'PATH':'/usr/bin:/bin','HOME':str(home),'MARKER':str(marker)},capture_output=True,text=True, encoding='utf-8', errors='replace',timeout=5) +assert marker.read_text(encoding='utf-8')=='executed' +print(json.dumps({'case':'newline safe execution','exit':p.returncode,'marker':marker.read_text(encoding='utf-8'),'stdout':p.stdout,'stderr':p.stderr})) +# Prove the benign backtick argument really is well-formed and matches input. +f=home/'f';f.write_text('needle\n', encoding='utf-8') +cmd='grep -e `printf needle` '+str(f) +p=subprocess.run(['/bin/bash','--noprofile','--norc','-c',cmd],env={'PATH':'/usr/bin:/bin','HOME':str(home)},capture_output=True,text=True, encoding='utf-8', errors='replace',timeout=5) +assert p.returncode==0 and p.stdout=='needle\n' +print(json.dumps({'case':'backtick argument','command':cmd,'baseline':(b.detect_hardline_command(cmd) if b else None),'head':d.detect_hardline_command(cmd),'tokens':d._shell_tokens_with_spans(cmd,0),'exit':p.returncode,'stdout':p.stdout})) +# Reporter's exact spelling, fixture makes the sed address meaningful. +with tempfile.TemporaryDirectory() as tmp: + Path(tmp,'f').write_text('X\ny\nz\nw\n', encoding='utf-8') + cmd='sed -n "$(grep -n X f | cut -d: -f1),+3p" f' + p=subprocess.run(['/bin/bash','--noprofile','--norc','-c',cmd],cwd=tmp,env={'PATH':'/usr/bin:/bin','HOME':str(home)},capture_output=True,text=True, encoding='utf-8', errors='replace',timeout=5) + assert p.returncode==0 and p.stdout=='X\ny\nz\nw\n' + print(json.dumps({'case':'reported real fixture','baseline':(b.detect_hardline_command(cmd) if b else None),'head':d.detect_hardline_command(cmd),'exit':p.returncode,'stdout':p.stdout})) diff --git a/evals/postmortem/run.py b/evals/postmortem/run.py new file mode 100644 index 0000000000..5d078e3ab0 --- /dev/null +++ b/evals/postmortem/run.py @@ -0,0 +1,87 @@ +#!/usr/bin/env python3 +"""Run the post-mortem harness against one or two checkouts and print a comparison table. + + python -m evals.postmortem.run --repo /path/to/checkout # one ref: pass/fail per probe + python -m evals.postmortem.run --repo A --compare B # two refs: side by side + python -m evals.postmortem.run --repo A --live # also the probes that spend money + +Each probe is a standalone script run in a fresh interpreter with the target checkout on sys.path and a +temp HERMES_HOME (probes that need real credentials say so and are only run with --live). A probe +"passes" when its process exits 0 AND its stdout contains the expected marker documented in +PROBES below; the marker is the behaviour the corresponding PR fixed. Run the forensics lanes +separately (they need a state.db copy): see forensics/README section in ../README.md. +""" +from __future__ import annotations + +import argparse +import os +import subprocess +import sys +import tempfile +from pathlib import Path + +HERE = Path(__file__).resolve().parent + +# (script, args-template, expected stdout substring, live?, PR) +# Two probes pass on main as well: scanner_bypass_probe (main blocked the witness too, as "malformed") and +# notice_delivery_probe (main has no interim notice to mis-deliver). They guard against regressing INTO the +# round-1 defects, which is why they are here. +PROBES = [ + ("live_ab/hardline_scanner_matrix.py", ["{repo}"], "ALL OK", False, "#103492"), + ("review_probes/scanner_bypass_probe.py", ["{repo}"], '"hardline": true', False, "#103492"), + ("live_ab/subagent_context_cap.py", ["{repo}"], "trigger=200,000", False, "#103513"), + ("live_ab/batch_failure_notice.py", ["{repo}"], "TASK_FAILURE_NOTICE", False, "#103549"), + ("review_probes/notice_delivery_probe.py", ["{repo}"], "PROBE_COMPLETE", False, "#103549"), + ("review_probes/cache_estimator_probe.py", ["{repo}"], '"preflight_should_compress": true', False, "#103476"), + ("review_probes/rewrite_hint_probe.py", ["{repo}"], "remote_fifo", False, "#103551"), + ("live_ab/auth_stampede.py", ["{repo}", "12"], "server_401=0", False, "#103526"), + ("review_probes/credential_identity_probe.py", ["{repo}", "pr"], '"after_sub": "account-A"', False, "#103526"), + # live (real provider calls, cents each) + ("live_ab/goal_judge_wait.py", ["{repo}", "3"], "('wait', ", True, "#103534"), + ("live_ab/cache_prefix_wire.py", ["{repo}", "B"], "", True, "#103476"), +] + + +def run_probe(script: str, args: list[str], repo: str, timeout: int = 240) -> tuple[int, str]: + env = dict(os.environ) + env.setdefault("HERMES_HOME", tempfile.mkdtemp(prefix="pm-probe-")) + env["PYTHONPATH"] = repo + os.pathsep + env.get("PYTHONPATH", "") + cmd = [sys.executable, str(HERE / script), *[a.format(repo=repo) for a in args]] + try: + p = subprocess.run(cmd, cwd=repo, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, env=env) + return p.returncode, (p.stdout + "\n" + p.stderr) + except subprocess.TimeoutExpired: + return 124, "TIMEOUT" + + +def main(argv=None) -> int: + ap = argparse.ArgumentParser(description=(__doc__ or "").split("\n\n")[0]) + ap.add_argument("--repo", required=True) + ap.add_argument("--compare", default=None, help="second checkout to run side by side (e.g. main vs branch)") + ap.add_argument("--live", action="store_true", help="also run probes that make real provider calls") + ap.add_argument("--only", default=None, help="substring filter on script path or PR number") + a = ap.parse_args(argv) + repos = [a.repo] + ([a.compare] if a.compare else []) + rows = [] + for script, args, marker, live, pr in PROBES: + if live and not a.live: + continue + if a.only and a.only not in script and a.only not in pr: + continue + cells = [] + for repo in repos: + rc, out = run_probe(script, args, os.path.abspath(repo)) + ok = rc == 0 and (marker in out if marker else True) + cells.append("PASS" if ok else f"FAIL(rc={rc})") + rows.append((pr, script, *cells)) + width = max(len(r[1]) for r in rows) if rows else 20 + head = f"{'PR':<9} {'probe':<{width}} " + " ".join(f"{os.path.basename(os.path.normpath(r)):<14}" for r in repos) + print(head); print("-" * len(head)) + for r in rows: + print(f"{r[0]:<9} {r[1]:<{width}} " + " ".join(f"{c:<14}" for c in r[2:])) + failed = any("FAIL" in c for r in rows for c in r[2 + (1 if a.compare else 0):]) # only the LAST column gates + return 1 if failed else 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/evals/postmortem/tests/__init__.py b/evals/postmortem/tests/__init__.py new file mode 100644 index 0000000000..e69de29bb2 diff --git a/evals/postmortem/tests/test_postmortem_harness.py b/evals/postmortem/tests/test_postmortem_harness.py new file mode 100644 index 0000000000..870e81cfae --- /dev/null +++ b/evals/postmortem/tests/test_postmortem_harness.py @@ -0,0 +1,88 @@ +"""The post-mortem forensics run end-to-end on a synthetic state.db and report the run population correctly. + +Guards the harness itself (it lives in evals/, outside the normal import graph): the root is discovered, +a compression-rollover child is excluded, pricing is fitted, and each lane writes its JSON without error. +""" +import json +import sqlite3 +import sys +import time +from pathlib import Path + +import pytest + +EVALS = Path(__file__).resolve().parents[2] + + +def _mk_db(path: Path) -> None: + c = sqlite3.connect(path) + c.executescript(""" + CREATE TABLE sessions(id TEXT PRIMARY KEY, parent_session_id TEXT, source TEXT, started_at REAL, ended_at REAL, + api_call_count INTEGER, input_tokens INTEGER, cache_read_tokens INTEGER, cache_write_tokens INTEGER, + output_tokens INTEGER, estimated_cost_usd REAL, system_prompt_hash TEXT); + CREATE TABLE system_prompts(hash TEXT PRIMARY KEY, prompt TEXT); + CREATE TABLE messages(id INTEGER PRIMARY KEY, session_id TEXT, role TEXT, content TEXT, tool_calls TEXT, + tool_name TEXT, reasoning TEXT, timestamp REAL); + CREATE TABLE state_meta(key TEXT PRIMARY KEY, value TEXT); + """) + t0 = time.time() - 4000 + c.execute("INSERT INTO system_prompts VALUES ('h1', ?)", ("x" * 35_000,)) + + def sess(sid, parent, source, start, end, calls, cr, cw, out): + cost = cr * 0.2e-6 + cw * 10e-6 + out * 40e-6 + c.execute("INSERT INTO sessions VALUES (?,?,?,?,?,?,?,?,?,?,?,?)", (sid, parent, source, start, end, calls, 0, cr, cw, out, cost, "h1")) + + sess("root", None, "cli", t0, t0 + 3600, 40, 2_000_000, 300_000, 40_000) + for i in range(12): # children at depth 1 + sess(f"c{i}", "root", "subagent", t0 + 10 + i, t0 + 600 + 60 * i, 20, 1_000_000, 200_000, 20_000) + sess("g0", "c0", "subagent", t0 + 20, t0 + 500, 10, 500_000, 100_000, 10_000) # depth 2 + sess("rollover", "root", "cli", t0 + 3601, t0 + 7200, 30, 9_000_000, 900_000, 90_000) # excluded + sess("unrelated", None, "telegram", t0, t0 + 100, 1, 1000, 100, 10) + # messages: a delegate_task timeout in c0, a truncated batch block in root, a hardline block, a nudge + c.execute("INSERT INTO messages(session_id,role,content,tool_name,timestamp) VALUES ('c0','tool',\"Error executing tool 'delegate_task': timed out after 420.0s\",'delegate_task',?)", (t0 + 100,)) + c.execute("INSERT INTO messages(session_id,role,content,tool_calls,timestamp) VALUES ('c0','assistant','','[{\"function\":{\"name\":\"terminal\",\"arguments\":\"{\\\\\"command\\\\\": \\\\\"sleep 600\\\\\"}\"}}]',?)", (t0 + 200,)) + c.execute("INSERT INTO messages(session_id,role,content,timestamp) VALUES ('root','user','[ASYNC DELEGATION BATCH COMPLETE — d]\\n--- ✓ TASK 1/2 ...\\n[SUMMARY TRUNCATED]\\n--- ✓ TASK 2/2 ...',?)", (t0 + 700,)) + c.execute("INSERT INTO messages(session_id,role,content,tool_name,timestamp) VALUES ('c1','tool','BLOCKED (hardline): command parser limit or malformed executable payload','terminal',?)", (t0 + 300,)) + c.execute("INSERT INTO messages(session_id,role,content,timestamp) VALUES ('root','assistant','Waiting on 3 batches; nothing to dispatch.',?)", (t0 + 800,)) + c.execute("INSERT INTO messages(session_id,role,content,timestamp) VALUES ('root','user','[Continuing toward your standing goal]\\nGoal: x',?)", (t0 + 810,)) + c.execute("INSERT INTO state_meta VALUES ('goal:root', ?)", (json.dumps({"status": "active", "waiting_on_session": "proc_x", "waiting_since": t0 + 900, "last_verdict": "wait", "turns_used": 3}),)) + c.commit(); c.close() + + +@pytest.fixture +def harness_path(monkeypatch): + monkeypatch.syspath_prepend(str(EVALS.parent)) + return EVALS + + +def test_run_population_excludes_rollover_and_unrelated_sessions(tmp_path, harness_path): + from evals.postmortem.forensics.common import Run + db = tmp_path / "state.db"; _mk_db(db) + run = Run.open(str(db), out=str(tmp_path / "out")) + assert run.root == "root" + assert set(run.in_run) == {"root", *{f"c{i}" for i in range(12)}, "g0"} + assert "rollover" not in run.in_run and "unrelated" not in run.in_run + s = run.summary() + assert s["by_depth"] == {0: 1, 1: 12, 2: 1} + assert abs(s["fitted_price_per_million"]["cache_write_tokens"] - 10.0) < 0.01 + assert abs(s["cost_usd"] - sum(run.cost(x) for x in run.in_run)) < 1e-6 + + +def test_every_lane_runs_and_writes_its_report(tmp_path, harness_path): + from evals.postmortem.forensics import delegation, goal_loop, tokens, tools + db = tmp_path / "state.db"; _mk_db(db) + out = tmp_path / "out" + for lane in (tokens, delegation, tools, goal_loop): + assert lane.main(["--db", str(db), "--out", str(out)]) == 0 + assert json.loads((out / "delegation.json").read_text(encoding="utf-8"))["observed"]["delegate_task_timeouts"] == 1 + assert json.loads((out / "tools.json").read_text(encoding="utf-8"))["observed"]["hardline_blocks_malformed_class"] == 1 + g = json.loads((out / "goal_loop.json").read_text(encoding="utf-8"))["observed"] + assert g["nudges"] == 1 and g["nudges_within_180s_of_a_waiting_turn"] == 1 + assert (out / "tokens.json").exists() + + +def test_runner_lists_a_probe_per_pr(harness_path): + from evals.postmortem import run as runner + prs = {p[4] for p in runner.PROBES} + assert {"#103492", "#103513", "#103549", "#103476", "#103551", "#103526", "#103534"} <= prs + assert sys.version_info >= (3, 10) From 463292351fec7ea74013e8ff6080eba82806cab3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 13:27:22 -0700 Subject: [PATCH 051/276] fix: mid-turn user message no longer waits behind a foreground terminal command MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A message typed while the agent runs (CLI busy_input_mode=interrupt, gateway priority redirect, ACP redirect) goes through AIAgent.redirect(). During tool execution redirect() degrades to steer(), whose delivery rides the tool result — so a long foreground command (a `sleep 285` CI poller, a build) parked the user's message until it exited. The UI printed "Redirected current turn" while nothing happened for minutes. redirect() now also asks the tool workers to YIELD (tools/interrupt.request_yield). The local terminal backend's wait loop honours it: the drain thread is stopped, the still-running Popen is adopted by the process registry as a notify_on_complete background session (ProcessRegistry.adopt_local — output so far seeds the buffer, the registry reader continues from the pipe), and the tool returns immediately with status "yielded_to_background" + session_id. The command is never killed; the completion notification arrives as usual and process(poll/wait/log/kill) work on it. Non-local backends and internal env.execute() consumers pass no yield_handler and are unaffected; a stale yield bit is cleared with the interrupt bit per worker tid. --- agent/interrupt_control.py | 15 +++- tests/tools/test_terminal_task_cwd.py | 6 +- .../test_terminal_yield_to_background.py | 88 +++++++++++++++++++ tools/environments/base.py | 32 +++++-- tools/environments/base_output.py | 19 ++-- tools/interrupt.py | 35 +++++++- tools/process_registry.py | 19 ++++ tools/terminal_tool.py | 20 ++++- tools/terminal_tool_background.py | 45 ++++++++++ website/docs/user-guide/cli.md | 4 +- website/docs/user-guide/messaging/index.md | 2 +- 11 files changed, 264 insertions(+), 21 deletions(-) create mode 100644 tests/tools/test_terminal_yield_to_background.py diff --git a/agent/interrupt_control.py b/agent/interrupt_control.py index e06f3ac13a..f99aca9dc1 100644 --- a/agent/interrupt_control.py +++ b/agent/interrupt_control.py @@ -9,6 +9,7 @@ import threading from typing import Optional from agent.interrupt_compat import request_hard_interrupt +from tools.interrupt import request_yield as _request_yield from tools.interrupt import set_interrupt as _set_interrupt # Same logger name as the origin module so log records / caplog filters are unchanged. @@ -245,8 +246,20 @@ class InterruptControlMixin: return False # Never kill a tool to deliver guidance; the steer drain puts it on the final tool result. + # A foreground terminal command would park that delivery until it exits (a 5-minute + # `sleep` poller, a build), so ask the tool workers to YIELD: terminal hands the live + # process to the background registry and returns; tools that don't yield are unaffected. if getattr(self, "_executing_tools", False): - return self.steer(cleaned) + accepted = self.steer(cleaned) + if accepted: + tracker = getattr(self, "_tool_worker_threads", None) + tracker_lock = getattr(self, "_tool_worker_threads_lock", None) + if tracker is not None and tracker_lock is not None: + with tracker_lock: + worker_tids = list(tracker) + for tid in worker_tids: + _request_yield(tid) + return accepted _model_active = getattr(self, "_model_request_active", None) with _ic_lock(self, "_pending_redirect_lock"): diff --git a/tests/tools/test_terminal_task_cwd.py b/tests/tools/test_terminal_task_cwd.py index 3246d5fa87..95ed40f5ed 100644 --- a/tests/tools/test_terminal_task_cwd.py +++ b/tests/tools/test_terminal_task_cwd.py @@ -40,7 +40,8 @@ def test_foreground_command_uses_registered_task_cwd_for_existing_environment(mo result = json.loads(terminal_tool.terminal_tool(command="pwd", task_id=task_id)) assert result["exit_code"] == 0 - assert calls == [("pwd", {"timeout": 60, "cwd": "/workspace/acp", "bounded_capture": True})] + assert len(calls) == 1 and calls[0][0] == "pwd" + assert calls[0][1] | {"timeout": 60, "cwd": "/workspace/acp", "bounded_capture": True} == calls[0][1] def test_explicit_workdir_still_wins_over_registered_task_cwd(monkeypatch): @@ -73,7 +74,8 @@ def test_explicit_workdir_still_wins_over_registered_task_cwd(monkeypatch): ) assert result["exit_code"] == 0 - assert calls == [{"timeout": 60, "cwd": "/explicit/workdir", "bounded_capture": True}] + assert len(calls) == 1 + assert calls[0] | {"timeout": 60, "cwd": "/explicit/workdir", "bounded_capture": True} == calls[0] def test_explicit_workdir_does_not_persist_into_session_cwd(monkeypatch): diff --git a/tests/tools/test_terminal_yield_to_background.py b/tests/tools/test_terminal_yield_to_background.py new file mode 100644 index 0000000000..1dec580af7 --- /dev/null +++ b/tests/tools/test_terminal_yield_to_background.py @@ -0,0 +1,88 @@ +"""A user message sent while a foreground terminal command runs must not wait for the command. + +``AIAgent.redirect()`` during tool execution degrades to ``steer()``, whose delivery rides +the tool result — so a long foreground command (a 5-minute ``sleep`` poller, a build) parked +the user's message until it exited. Now redirect() also asks the tool workers to YIELD: the +local terminal backend hands the still-running process to the process registry as a +notify-on-complete background session and returns immediately, without killing it. +""" +import json +import os +import threading +import time + +import pytest + +from agent.interrupt_control import InterruptControlMixin +from tools import interrupt as interrupt_mod +from tools.process_registry import process_registry +from tools.terminal_tool import terminal_tool + +pytestmark = pytest.mark.linux_only + + +class _Agent(InterruptControlMixin): + _executing_tools = True + _interrupt_requested = False + _pending_steer = None + _pending_redirect = None + api_mode = "chat_completions" + + def __init__(self): + self._pending_steer_lock = threading.Lock() + self._pending_redirect_lock = threading.Lock() + self._tool_worker_threads = set() + self._tool_worker_threads_lock = threading.Lock() + self._execution_thread_id = None + + +def test_redirect_mid_command_yields_it_to_background_without_killing_it(tmp_path, monkeypatch): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + agent = _Agent() + res = {} + + def worker(): + with agent._tool_worker_threads_lock: + agent._tool_worker_threads.add(threading.current_thread().ident) + t0 = time.monotonic() + res["result"] = json.loads(terminal_tool("echo started; sleep 60; echo done", task_id="yield-test", timeout=90)) + res["elapsed"] = time.monotonic() - t0 + + t = threading.Thread(target=worker, daemon=True) + t.start() + time.sleep(1.5) + assert agent.redirect("also, check the session ids") is True + assert agent._pending_steer == "also, check the session ids" # still delivered as a steer + + t.join(timeout=15) + assert not t.is_alive(), "terminal tool still blocked on the command after redirect()" + r = res["result"] + try: + assert r["status"] == "yielded_to_background" + assert r["exit_code"] is None and "started" in r["output"] + assert r["notify_on_complete"] is True + # The process is alive and tracked: poll/wait/kill and the completion notification work. + assert os.path.exists(f"/proc/{r['pid']}") + assert process_registry.poll(r["session_id"])["status"] == "running" + assert not interrupt_mod.is_thread_yield_requested(t.ident) + finally: + killed = process_registry.kill_process(r["session_id"]) + assert killed["status"] == "killed" + evt = process_registry.completion_queue.get(timeout=5) + assert evt["session_id"] == r["session_id"] + + +def test_yield_request_without_steer_leaves_foreground_wait_alone(): + """No yield handler (internal env.execute consumers) -> the request is ignored and the + command runs to completion; an unrelated stale yield bit must not leak into it either.""" + from tools.environments.local import LocalEnvironment + + env = LocalEnvironment() + try: + interrupt_mod.request_yield(threading.current_thread().ident) + result = env.execute("echo alpha; sleep 0.3; echo omega", timeout=10) + assert result["returncode"] == 0 + assert "omega" in result["output"] + finally: + interrupt_mod.consume_yield(threading.current_thread().ident) + env.cleanup() diff --git a/tools/environments/base.py b/tools/environments/base.py index 3a19d63246..47998cd27b 100644 --- a/tools/environments/base.py +++ b/tools/environments/base.py @@ -19,7 +19,7 @@ from pathlib import Path from typing import Callable, Iterable from hermes_constants import get_hermes_home -from tools.interrupt import is_interrupted, is_thread_interrupted +from tools.interrupt import consume_yield, is_interrupted, is_thread_interrupted from tools.environments.base_output import ( ProcessHandle, _finalize_wait_result, _new_output_collector, _start_drain_thread, ) @@ -337,8 +337,13 @@ class BaseEnvironment(ABC): # --- Process lifecycle --- def _wait_for_process( self, proc: ProcessHandle, timeout: int = 120, *, - bounded_capture: bool = False, watch_interrupt_tid: int | None = None) -> dict: + bounded_capture: bool = False, watch_interrupt_tid: int | None = None, + yield_handler: Callable[[ProcessHandle, str], dict] | None = None) -> dict: """Poll-based wait with interrupt checking and stdout draining (shared, not overridden). + ``yield_handler(proc, output_so_far)``: when the tool thread is asked to yield + (``tools.interrupt.request_yield`` — a user message arrived mid-command), the drain + thread is stopped, the still-running process is handed to the handler and its dict + is returned as the result; the process is NOT killed. ``bounded_capture=True`` (foreground terminal-tool path only) retains at most ``tool_output.max_bytes`` in a head/tail window so a verbose subprocess cannot OOM the process; the default keeps full fidelity for internal consumers. Fires the activity @@ -354,7 +359,8 @@ class BaseEnvironment(ABC): data. See #64435. """ output = _new_output_collector(proc, bounded_capture) - drain_thread = _start_drain_thread(proc, output) + drain_stop = threading.Event() if yield_handler is not None else None + drain_thread = _start_drain_thread(proc, output, drain_stop) _now = time.monotonic() deadline = _now + timeout _activity_state = {"last_touch": _now, "start": _now} @@ -375,6 +381,19 @@ class BaseEnvironment(ABC): trace.interrupted() _kill_and_join() return self._finalize_wait_result(output, output.render(suffix="\n[Command interrupted]"), 130) + if yield_handler is not None and consume_yield(watch_interrupt_tid): + drain_stop.set() + drain_thread.join(timeout=1) + try: + handed = yield_handler(proc, output.render()) + except Exception: + logger.warning("yield-to-background handoff failed; continuing to wait", exc_info=True) + handed = None + if handed is not None: + output.close_spill() + return handed + drain_stop.clear() + drain_thread = _start_drain_thread(proc, output, drain_stop) if time.monotonic() > deadline: trace.timed_out() _kill_and_join() @@ -462,7 +481,8 @@ class BaseEnvironment(ABC): timeout: int | None = None, stdin_data: str | None = None, rewrite_compound_background: bool = True, - bounded_capture: bool = False) -> dict: + bounded_capture: bool = False, + yield_handler: Callable[[ProcessHandle, str], dict] | None = None) -> dict: """Execute a command, return {"output": str, "returncode": int}. ``bounded_capture=True`` caps retention at ``tool_output.max_bytes`` WHILE draining; only the foreground terminal tool may set it — internal full-fidelity consumers (file-op ``cat`` reads feeding the @@ -509,7 +529,9 @@ class BaseEnvironment(ABC): spawned = self._run_bash(wrapped, login=login, timeout=effective_timeout, stdin_data=effective_stdin) proc_holder.append(spawned) return self._wait_for_process( - spawned, timeout=effective_timeout, bounded_capture=bounded_capture, watch_interrupt_tid=parent_tid) + spawned, timeout=effective_timeout, bounded_capture=bounded_capture, + watch_interrupt_tid=parent_tid, + **({"yield_handler": yield_handler} if yield_handler is not None else {})) def _on_timeout() -> None: if proc_holder: diff --git a/tools/environments/base_output.py b/tools/environments/base_output.py index 48350450a1..e8749b3c60 100644 --- a/tools/environments/base_output.py +++ b/tools/environments/base_output.py @@ -341,7 +341,7 @@ class _ThreadedProcessHandle: # --- Stdout drain thread --- -def _drain_stdout(proc: ProcessHandle, output: _BoundedOutputCollector) -> None: +def _drain_stdout(proc: ProcessHandle, output: _BoundedOutputCollector, stop: "threading.Event | None" = None) -> None: """Drain ``proc.stdout`` into *output* until EOF or shortly after exit. ``for line in proc.stdout`` would block on ``readline()`` until EOF, and a backgrounded grandchild (``cmd &``, ``setsid cmd & disown``) inherits the pipe's write end — so the @@ -384,7 +384,7 @@ def _drain_stdout(proc: ProcessHandle, output: _BoundedOutputCollector) -> None: while chunk := os.read(fd, 4096): output.append(decoder.decode(chunk)) else: - _drain_fd_select(proc, fd, output, decoder) + _drain_fd_select(proc, fd, output, decoder, stop) except Exception: pass # closed fd / broken stream: keep what was captured finally: @@ -397,10 +397,13 @@ def _drain_stdout(proc: ProcessHandle, output: _BoundedOutputCollector) -> None: pass -def _drain_fd_select(proc, fd: int, output: _BoundedOutputCollector, decoder) -> None: - """POSIX drain: select() poll, stopping ~300ms after bash exits with the pipe idle.""" +def _drain_fd_select(proc, fd: int, output: _BoundedOutputCollector, decoder, stop=None) -> None: + """POSIX drain: select() poll, stopping ~300ms after bash exits with the pipe idle, or + when *stop* is set (the pipe is being handed to another reader — yield-to-background).""" idle_after_exit = 0 while True: + if stop is not None and stop.is_set(): + return try: ready, _, _ = select.select([fd], [], [], 0.1) except (ValueError, OSError): @@ -422,8 +425,10 @@ def _drain_fd_select(proc, fd: int, output: _BoundedOutputCollector, decoder) -> return -def _start_drain_thread(proc: ProcessHandle, output: _BoundedOutputCollector) -> threading.Thread: - """Start the daemon thread running :func:`_drain_stdout`.""" - thread = threading.Thread(target=_drain_stdout, args=(proc, output), daemon=True) +def _start_drain_thread( + proc: ProcessHandle, output: _BoundedOutputCollector, stop: "threading.Event | None" = None, +) -> threading.Thread: + """Start the daemon thread running :func:`_drain_stdout`; *stop* ends it early.""" + thread = threading.Thread(target=_drain_stdout, args=(proc, output, stop), daemon=True) thread.start() return thread diff --git a/tools/interrupt.py b/tools/interrupt.py index 99718e0438..e9ca0b994e 100644 --- a/tools/interrupt.py +++ b/tools/interrupt.py @@ -20,12 +20,15 @@ if _DEBUG_INTERRUPT: # Interrupted thread idents + optional user-safe cause (never the user's message text). _interrupted_threads: set[int] = set() _interrupt_reasons: dict[int, str] = {} +# Threads asked to YIELD: hand a long-running foreground command to the background +# instead of killing it, so a mid-turn user message is not parked behind it. +_yield_threads: set[int] = set() _lock = threading.Lock() def set_interrupt(active: bool, thread_id: int | None = None, *, reason: str | None = None) -> None: """Set or clear the interrupt for *thread_id* (default: current thread); ``reason`` is - an optional user-safe cause.""" + an optional user-safe cause. Clearing also drops a pending yield request.""" tid = thread_id if thread_id is not None else threading.current_thread().ident with _lock: (_interrupted_threads.add if active else _interrupted_threads.discard)(tid) @@ -33,6 +36,8 @@ def set_interrupt(active: bool, thread_id: int | None = None, *, reason: str | N _interrupt_reasons[tid] = reason else: _interrupt_reasons.pop(tid, None) + if not active: + _yield_threads.discard(tid) _snapshot = set(_interrupted_threads) if _DEBUG_INTERRUPT else None if _DEBUG_INTERRUPT: logger.info( @@ -58,6 +63,34 @@ def is_thread_interrupted(thread_id: int | None) -> bool: return thread_id in _interrupted_threads +def request_yield(thread_id: int) -> None: + """Ask the tool running on *thread_id* to yield: a foreground terminal command hands + its live process to the background registry and returns at once, so a user's mid-turn + message (``redirect()`` during tool execution) is delivered instead of parked behind it. + The command itself is never killed; that is what ``set_interrupt`` is for.""" + with _lock: + _yield_threads.add(thread_id) + + +def is_thread_yield_requested(thread_id: int | None) -> bool: + """Whether a yield is pending for *thread_id* (``None`` never is).""" + if thread_id is None: + return False + with _lock: + return thread_id in _yield_threads + + +def consume_yield(thread_id: int | None) -> bool: + """Atomically take the pending yield for *thread_id*; True if one was pending.""" + if thread_id is None: + return False + with _lock: + if thread_id in _yield_threads: + _yield_threads.discard(thread_id) + return True + return False + + def run_if_not_interrupted(callback: Callable[[], None]) -> bool: """Run a state transition atomically with current-thread interruption. diff --git a/tools/process_registry.py b/tools/process_registry.py index 75ae003b84..afdd52a0e7 100644 --- a/tools/process_registry.py +++ b/tools/process_registry.py @@ -890,6 +890,25 @@ class ProcessRegistry: with suppress(Exception): proc.wait(timeout=5) + def adopt_local( + self, proc: subprocess.Popen, *, command: str, cwd: Optional[str], task_id: str = "", + session_key: str = "", owner_task_id: str = "", output_so_far: str = "", + notify_on_complete: bool = True) -> ProcessSession: + """Take over a still-running foreground Popen as a tracked background session + (yield-to-background: the user sent a message while the command was running). + The caller has stopped its own drain thread; the registry's reader continues from + the pipe's current position and ``output_so_far`` seeds the buffer so nothing + already captured is lost.""" + session = self._new_session(command, task_id, owner_task_id, session_key, cwd) + session.process = proc + session.pid = proc.pid + session.host_start_time = self._safe_host_start_time(session.pid) + session.notify_on_complete = notify_on_complete + if output_so_far: + session.append_output(output_so_far) + self._track_started(session, self._reader_loop, f"proc-reader-{session.id}") + return session + def spawn_via_env( self, env: Any, command: str, cwd: str = None, task_id: str = "", session_key: str = "", timeout: int = 10, owner_task_id: str = "") -> ProcessSession: diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index 713c5bdf23..9ee4eca08b 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -736,7 +736,7 @@ from tools.terminal_tool_guards import ( _foreground_background_guidance, _safe_command_preview, _validate_workdir, gateway_lifecycle_block, self_repo_block, ) -from tools.terminal_tool_background import spawn_background_process +from tools.terminal_tool_background import _YIELDED_NOTE, spawn_background_process, yield_to_background_handler from tools.terminal_tool_result import finalize_foreground_result @@ -1009,6 +1009,12 @@ def _acquire_env(plan: _ExecPlan, task_id: Optional[str]) -> Any: return new_env +def _yield_kwargs(command: str, **ctx) -> dict: + """``env.execute`` kwargs enabling yield-to-background (local backend only).""" + handler = yield_to_background_handler(command=command, **ctx) + return {"yield_handler": handler} if handler is not None else {} + + def _run_foreground( command: str, env: Any, plan: _ExecPlan, *, task_id: Optional[str], session_id: Optional[str], session_key: str, @@ -1035,7 +1041,11 @@ def _run_foreground( # bounded_capture: model-facing output keeps a head/tail window # while streaming so a verbose command can't OOM the gateway; # internal env.execute() consumers stay unbounded. - result = env.execute(command, timeout=effective_timeout, cwd=command_cwd, bounded_capture=True) + result = env.execute( + command, timeout=effective_timeout, cwd=command_cwd, bounded_capture=True, + **_yield_kwargs(command, env_type=env_type, cwd=command_cwd, effective_task_id=eff, + task_id=task_id, session_key=session_key), + ) break except Exception as e: if "timeout" in str(e).lower(): @@ -1051,6 +1061,12 @@ def _run_foreground( max_retries, _safe_command_preview(command), type(e).__name__, e, eff, env_type) return _error_json(_redact_terminal_error_text(f"Command execution failed: {type(e).__name__}: {e}")) + if result.get("yielded_session_id"): + return json.dumps({ + "output": result.get("output", ""), "exit_code": None, "error": None, + "status": "yielded_to_background", "session_id": result["yielded_session_id"], + "pid": result.get("pid"), "notify_on_complete": True, "note": _YIELDED_NOTE, + }, ensure_ascii=False) return finalize_foreground_result( command=command, result=result, env=env, env_type=env_type, effective_task_id=eff, task_id=task_id, session_id=session_id, session_key=session_key, workdir=workdir, diff --git a/tools/terminal_tool_background.py b/tools/terminal_tool_background.py index f9afd34532..d611b17c39 100644 --- a/tools/terminal_tool_background.py +++ b/tools/terminal_tool_background.py @@ -186,3 +186,48 @@ def spawn_background_process( "output": "", "exit_code": -1, "error": _redact_terminal_error_text(f"Failed to start background process: {e}"), }, ensure_ascii=False) + + +_YIELDED_NOTE = ( + "The user sent a message while this command was running, so it was moved to the " + "background WITHOUT being killed and is still running. You will be notified when it " + "exits (notify_on_complete). Read the user's message and respond to it now; use " + "process(action='poll'|'wait'|'log', session_id=...) to check on this command." +) + + +def yield_to_background_handler( + *, command: str, env_type: str, cwd: Optional[str], effective_task_id: str, + task_id: Optional[str], session_key: str, +): + """Build the ``yield_handler`` a foreground ``env.execute`` calls when the tool thread is + asked to yield (a user message arrived mid-command). Local backend only: the live Popen + is adopted by the process registry as a notify-on-complete background session and the + partial output is returned to the model right away. Other backends return None (no + adoptable host process) and the foreground wait continues.""" + if env_type != "local": + return None + + def _handler(proc, output_so_far: str) -> dict: + from tools.process_registry import process_registry + session = process_registry.adopt_local( + proc, command=command, cwd=cwd, task_id=effective_task_id, + owner_task_id=task_id or effective_task_id, session_key=session_key, + output_so_far=output_so_far) + _stamp_routing_if_gateway(process_registry, session, session_key) + logger.info("foreground command yielded to background as %s (pid %s)", session.id, session.pid) + return { + "output": output_so_far, "returncode": None, "yielded_session_id": session.id, "pid": session.pid, + } + return _handler + + +def _stamp_routing_if_gateway(process_registry, session, session_key) -> None: + """Route the adopted session's completion like a normal notify_on_complete spawn.""" + from gateway.session_context import async_delivery_supported, get_session_env + if not async_delivery_supported(): + session.notify_on_complete = False + return + _stamp_gateway_routing(session, get_session_env) + if session.watcher_platform: + _register_completion_watcher(process_registry, session, session_key) diff --git a/website/docs/user-guide/cli.md b/website/docs/user-guide/cli.md index 756038b8ba..a0176b8ec9 100644 --- a/website/docs/user-guide/cli.md +++ b/website/docs/user-guide/cli.md @@ -364,7 +364,7 @@ The `display.busy_input_mode` config key controls what happens when you press En | Mode | Behavior | |------|----------| -| `"interrupt"` (default) | Your message redirects the active turn. Model generation restarts with displayed reasoning and completed work preserved; running tools finish first | +| `"interrupt"` (default) | Your message redirects the active turn. Model generation restarts with displayed reasoning and completed work preserved. A running foreground terminal command is moved to the background (not killed — you get a completion notification) so your message is read immediately; other running tools finish first | | `"queue"` | Your message is silently queued and sent as the next turn after the agent finishes | | `"steer"` | Your message is injected into the current run via `/steer`, arriving at the agent after the next tool call — no interrupt, no new turn | @@ -374,7 +374,7 @@ display: busy_input_mode: "steer" # or "queue" or "interrupt" (default) ``` -`"queue"` mode prepares a separate follow-up turn. `"steer"` always waits for the next tool-result boundary. The default `"interrupt"` mode responds sooner during model generation while avoiding cancellation of a running tool. Use `/stop` when you want to cancel the turn and its foreground work. Unknown values fall back to `"interrupt"`. +`"queue"` mode prepares a separate follow-up turn. `"steer"` always waits for the next tool-result boundary. The default `"interrupt"` mode responds sooner during model generation while avoiding cancellation of a running tool; a long foreground `terminal` command (a build, a poller) is handed to the background so the agent sees your message right away instead of after the command exits. Use `/stop` when you want to cancel the turn and its foreground work. Unknown values fall back to `"interrupt"`. `"steer"` has two automatic fallbacks: if the agent hasn't started yet, or if images are attached, the message falls back to `"queue"` behavior so nothing is lost. diff --git a/website/docs/user-guide/messaging/index.md b/website/docs/user-guide/messaging/index.md index 72fc4288eb..a4cfe2592d 100644 --- a/website/docs/user-guide/messaging/index.md +++ b/website/docs/user-guide/messaging/index.md @@ -405,7 +405,7 @@ Send a message while the agent is working to correct the active turn: ### Queue vs interrupt vs steer (busy-input mode) -By default, messaging a busy agent redirects its active turn. Two other modes are available: +By default, messaging a busy agent redirects its active turn (a running foreground terminal command is moved to the background rather than killed, so your message is read immediately). Two other modes are available: - `queue` — follow-up messages wait and run as the next turn after the current task finishes. - `steer` — follow-up messages are injected into the current run via `/steer`, arriving at the agent after the next tool call. No interrupt, no new turn. Falls back to `queue` behavior if the agent hasn't started yet. From 4dc9988f7870b1a02e3987d6cd6ec9f93364273c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 11:11:23 -0700 Subject: [PATCH 052/276] feat(skills): reddit-reading and rss-feeds bundled skills, multi-platform sweep guidance Two zero-install research skills plus routing guidance, ported as ideas (not code) from the Agent Reach skill's per-platform backend routing. reddit-reading (skills/social-media): subreddit listings, site/subreddit search, threads with comments, user pages. Live-verified from a datacentre IP: www .json, api.reddit.com, old.reddit, r.jina.ai and the browser tool all return 403 / an empty shell / a humanity check; the Atom .rss endpoints are the only anonymous path and are throttled to ~1 request/min/IP. The script waits out the x-ratelimit-reset window once and retries, and switches to the OAuth API (scores, nested comments, ~100 req/min) when REDDIT_CLIENT_ID/SECRET are present. `doctor` reports the active backend. rss-feeds (skills/research): RSS 2.0 / RSS 1.0 / Atom / JSON Feed parsing with UTC-normalised dates, --since/--limit, and feed discovery behind a page URL (, then well-known paths). Stdlib only; the optional blogwatcher skill remains the stateful many-feed reader and now points at this one for one-off reads. grounded-citations gains a "Multi-Platform Sweeps" section routing "what are people saying about X" tasks across web, Reddit, feeds, video, code and X with per-platform attribution and coverage-gap reporting; competitor-news-monitor references the new sources. Tests: two invariant tests per skill (format normalisation + discovery; anonymous 429 handling + OAuth routing/flattening), no network. The reddit test caught a real bug: the feed footer's "/u/author" leaked into post bodies. --- optional-skills/research/blogwatcher/SKILL.md | 1 + .../research/competitor-news-monitor/SKILL.md | 4 +- skills/research/grounded-citations/SKILL.md | 25 +- skills/research/rss-feeds/SKILL.md | 96 ++++++ skills/research/rss-feeds/scripts/feed.py | 246 ++++++++++++++ skills/social-media/reddit-reading/SKILL.md | 107 ++++++ .../reddit-reading/scripts/reddit.py | 307 ++++++++++++++++++ tests/skills/test_reddit_reading_skill.py | 113 +++++++ tests/skills/test_rss_feeds_skill.py | 69 ++++ website/docs/reference/skills-catalog.md | 2 + .../research-competitor-news-monitor.md | 6 +- .../research/research-grounded-citations.md | 27 +- .../bundled/research/research-rss-feeds.md | 114 +++++++ .../social-media-reddit-reading.md | 125 +++++++ .../optional/research/research-blogwatcher.md | 3 +- website/sidebars.ts | 2 + 16 files changed, 1236 insertions(+), 11 deletions(-) create mode 100644 skills/research/rss-feeds/SKILL.md create mode 100644 skills/research/rss-feeds/scripts/feed.py create mode 100644 skills/social-media/reddit-reading/SKILL.md create mode 100644 skills/social-media/reddit-reading/scripts/reddit.py create mode 100644 tests/skills/test_reddit_reading_skill.py create mode 100644 tests/skills/test_rss_feeds_skill.py create mode 100644 website/docs/user-guide/skills/bundled/research/research-rss-feeds.md create mode 100644 website/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading.md diff --git a/optional-skills/research/blogwatcher/SKILL.md b/optional-skills/research/blogwatcher/SKILL.md index cf4b58472a..edd2449b01 100644 --- a/optional-skills/research/blogwatcher/SKILL.md +++ b/optional-skills/research/blogwatcher/SKILL.md @@ -24,6 +24,7 @@ Track blog and RSS/Atom feed updates with the `blogwatcher-cli` tool. Supports a - **Recurring watch — use the cronjob tool's `monitor` field, not a bare schedule.** `monitor` runs a script each tick and only wakes the agent when output changes: set it to a script that runs `blogwatcher-cli scan >/dev/null 2>&1 && blogwatcher-cli articles` (deterministic output; new articles = changed output = agent wakes with the diff injected). Unchanged ticks cost zero LLM calls. Set `deliver` to route digests to a chat/channel; add `continuity: true` so consecutive digests can dedupe. - **Reading an article the user asks about**: `web_extract([url])` on the article URL from `blogwatcher-cli articles` — do not re-scrape by hand. - **One-off "watch this page for changes" without feed semantics**: skip this skill; the cronjob tool's `monitor` field accepts an http(s) URL directly. +- **One-off read of a feed or a site's latest posts, nothing to install**: the bundled `rss-feeds` skill (`scripts/feed.py read URL`); blogwatcher earns its install when you track many feeds with read/unread state. - **Company/competitor tracking with analysis and citations**: prefer the `competitor-news-monitor` skill; blogwatcher is the lighter raw-feed layer it can sit on. ## Installation diff --git a/skills/research/competitor-news-monitor/SKILL.md b/skills/research/competitor-news-monitor/SKILL.md index c480f2b18b..e1fb7ef2d8 100644 --- a/skills/research/competitor-news-monitor/SKILL.md +++ b/skills/research/competitor-news-monitor/SKILL.md @@ -8,7 +8,7 @@ platforms: [linux, macos, windows] metadata: hermes: tags: [Competitors, News, Market-Research, Monitoring] - related_skills: [blogwatcher] + related_skills: [blogwatcher, rss-feeds, reddit-reading] --- # Competitor News Monitor @@ -42,7 +42,7 @@ For each company include, where available: 5. reputable trade and financial press 6. job postings as weak supporting evidence -Use `blogwatcher` for feeds and `web_search`/`web_extract` for pages. Write the watch contract (watchlist, categories, materiality threshold, last cutoff) to a state file under `~/.hermes/competitor-watches/.json`, then create the job: +Use `rss-feeds` (bundled) or `blogwatcher` (optional, stateful) for feeds, `reddit-reading` for community discussion, and `web_search`/`web_extract` for pages. Write the watch contract (watchlist, categories, materiality threshold, last cutoff) to a state file under `~/.hermes/competitor-watches/.json`, then create the job: ``` cronjob(action="create", diff --git a/skills/research/grounded-citations/SKILL.md b/skills/research/grounded-citations/SKILL.md index ae69fb98b0..599fc6b1d7 100644 --- a/skills/research/grounded-citations/SKILL.md +++ b/skills/research/grounded-citations/SKILL.md @@ -1,7 +1,7 @@ --- name: grounded-citations description: "Ground answers and documents in cited, verifiable sources." -version: 1.1.0 +version: 1.2.0 author: Hermes Agent + Teknium license: MIT platforms: [linux, macos, windows] @@ -9,7 +9,7 @@ metadata: hermes: tags: [Research, Citations, Grounding, Sources, Web, Reports] category: research - related_skills: [arxiv, pdf] + related_skills: [arxiv, pdf, reddit-reading, rss-feeds, youtube-content] --- # Grounded Citations @@ -126,6 +126,27 @@ sources, cite inline, end with the rendered `Sources:` list. For a short answer you may render the block from `sources.py render --only ` instead of writing to a file. +## Multi-Platform Sweeps + +"What are people saying about X" / "research X across the web" is not one +`web_search`. Fan out across source types, collect in parallel, then synthesise +with every claim attributed to the platform it came from: + +| Source type | Route | What it adds | +|---|---|---| +| Open web | `web_search` → `web_extract` | official docs, articles, announcements | +| Community discussion | `reddit-reading` (`search`, `thread`) | real user experience, complaints, workarounds | +| Blogs / releases / changelogs | `rss-feeds` (`read`, `discover`) | dated primary posts, version history | +| Video | `youtube-content` | walkthroughs, demos, talks | +| Code | `terminal` with `gh search repos` / `gh search issues` | implementations, open bugs | +| X/Twitter | `xurl` (needs API access) | announcements, developer chatter | + +Register every URL from every route in the ledger as it arrives (step ②). Keep +opinion and measurement apart: a Reddit thread is evidence that users *report* +something, not that it is true; pair it with a primary source or label it as +sentiment. Report per-platform coverage gaps ("Reddit search returned nothing +newer than March") rather than silently narrowing to what worked. + ## Fact-Checking Mode For work where the reader must be able to check the chain — medical, legal, diff --git a/skills/research/rss-feeds/SKILL.md b/skills/research/rss-feeds/SKILL.md new file mode 100644 index 0000000000..48033d4028 --- /dev/null +++ b/skills/research/rss-feeds/SKILL.md @@ -0,0 +1,96 @@ +--- +name: rss-feeds +description: "Read RSS, Atom, JSON feeds; discover feeds behind a page." +version: 1.0.0 +author: Teknium (teknium1), Hermes Agent +license: MIT +platforms: [linux, macos, windows] +metadata: + hermes: + tags: [RSS, Atom, Feeds, Monitoring, Research, Blogs, Releases] + related_skills: [reddit-reading, competitor-news-monitor, grounded-citations, youtube-content, blogwatcher] +--- + +# RSS Feeds Skill + +Reads any RSS 2.0, RSS 1.0/RDF, Atom, or JSON Feed URL into a clean, date-sorted list of +entries, and discovers the feed behind an ordinary page URL (`` or +the usual `/feed`, `/rss.xml`, `/atom.xml` paths). Standard library only, nothing to +install. It does not fetch full article bodies — pass an entry's link to `web_extract` for +that. + +## When to Use + +- "What's new on ", "latest releases of ", "recent posts in + ", "read this feed", "does this site have an RSS feed". +- Building a recurring digest with `cronjob_manage` (feeds are cheaper and more stable than + scraping the HTML front page every run). For a persistent read/unread database across + many feeds install the optional `blogwatcher` skill; this skill is the zero-install read. +- Anything where a structured list of `title / link / date / author / summary` beats a + rendered page: podcasts, changelogs, YouTube channels, newsrooms, forum categories. + +## Prerequisites + +None. Python 3.10+, network access to the feed host. + +## How to Run + +Run through `terminal` with the skill-relative script path: + +```bash +python3 scripts/feed.py read https://hnrss.org/frontpage --limit 10 +python3 scripts/feed.py read https://simonwillison.net/ # page URL → discovers the feed +python3 scripts/feed.py read URL --since 2026-09-01 --json # only newer entries, machine-readable +python3 scripts/feed.py discover https://example.com/ # list candidate feed URLs +``` + +## Quick Reference + +| Source | Feed URL pattern | +|---|---| +| GitHub releases / commits / tags | `https://github.com/OWNER/REPO/releases.atom`, `…/commits/BRANCH.atom`, `…/tags.atom` | +| Subreddit / Reddit search | `https://www.reddit.com/r/NAME/.rss`, `https://www.reddit.com/search.rss?q=…` (1 req/min anon; see `reddit-reading`) | +| YouTube channel | `https://www.youtube.com/feeds/videos.xml?channel_id=UC…` | +| Hacker News | `https://hnrss.org/frontpage`, `https://hnrss.org/newest?q=TERM` | +| arXiv category | `https://rss.arxiv.org/rss/cs.CL` | +| Substack / Medium / WordPress / Ghost | `SITE/feed`, `medium.com/feed/@user`, `SITE/rss/` | +| Podcasts | the show's RSS URL from its hosting page (`discover` finds it) | + +Output fields per entry: `title`, `link`, `published` (UTC ISO 8601), `author`, `summary` +(HTML stripped, ≤ 2000 chars). Entries are sorted newest-first. + +## Procedure + +① If you only have a site URL, run `read` on it directly; the script discovers the feed +and reports which URL it used (`discovered_from`). Use `discover` when you want to choose +between several advertised feeds (comments feed vs posts feed, per-category feeds). + +② Bound the request: `--limit` for "latest N", `--since YYYY-MM-DD` for "since last +check". For a cron digest persist the last-seen `published` value and pass it as +`--since` next run. + +③ For full text, hand the entry `link` to `web_extract`; feed summaries are frequently +truncated or the first paragraph only. + +④ Cite the entry `link`, not the feed URL, when the result feeds a report +(`grounded-citations`). + +## Pitfalls + +- A 200 response with HTML means the URL is a page, not a feed; the script falls through + to discovery automatically, but a site with no `` and none of the + common paths reports `no feed found` — check the site's footer or `/sitemap.xml` before + concluding there is none. +- Reddit feeds share Reddit's anonymous throttle (about one request per minute per IP). + Chain them through `reddit-reading`, which waits out the window, when you need more + than one Reddit call. +- Dates: RSS `pubDate` is RFC 822 and Atom uses ISO 8601; the script normalises both to + UTC. Feeds that omit dates sort to the bottom and are dropped by `--since`. +- Some feeds are Cloudflare-fronted and 403 non-browser clients; `blocked-page-recovery` + handles that class. + +## Verification + +`python3 scripts/feed.py read https://github.com/NousResearch/hermes-agent/releases.atom +--limit 1` prints one entry with a `releases/tag/` link and a `[atom]` format tag; +`discover https://simonwillison.net/` prints an `/atom/` URL. diff --git a/skills/research/rss-feeds/scripts/feed.py b/skills/research/rss-feeds/scripts/feed.py new file mode 100644 index 0000000000..24edef11f1 --- /dev/null +++ b/skills/research/rss-feeds/scripts/feed.py @@ -0,0 +1,246 @@ +#!/usr/bin/env python3 +"""Read RSS / Atom / JSON Feed sources and discover feeds behind a page URL. + +Standard library only, so it runs in any Hermes environment without an install +step. Output is JSON (``--json``) or a compact text listing. + + python3 feed.py read https://example.com/feed.xml [--limit N] [--since 2026-09-01] + python3 feed.py discover https://example.com/ + python3 feed.py read https://example.com/ -> discovers, then reads the first feed +""" + +from __future__ import annotations + +import argparse +import html +import json +import re +import sys +import urllib.error +import urllib.parse +import urllib.request +import xml.etree.ElementTree as ET +from datetime import datetime, timezone +from email.utils import parsedate_to_datetime + +USER_AGENT = "hermes-agent/1.0 (rss-feeds skill; +https://github.com/NousResearch/hermes-agent)" +TIMEOUT = 20 +NS = { + "atom": "http://www.w3.org/2005/Atom", + "dc": "http://purl.org/dc/elements/1.1/", + "content": "http://purl.org/rss/1.0/modules/content/", + "media": "http://search.yahoo.com/mrss/", +} +FEED_TYPES = ("application/rss+xml", "application/atom+xml", "application/feed+json", "application/json") +COMMON_FEED_PATHS = ("/feed", "/feed.xml", "/rss", "/rss.xml", "/atom.xml", "/index.xml", "/feed.json", "/blog/feed", "/blog/rss.xml") +_TAG_RE = re.compile(r"<[^>]+>") +_WS_RE = re.compile(r"\s+") + + +def fetch(url: str) -> tuple[bytes, str]: + req = urllib.request.Request(url, headers={"User-Agent": USER_AGENT, "Accept": "*/*"}) + with urllib.request.urlopen(req, timeout=TIMEOUT) as resp: + return resp.read(), resp.headers.get("Content-Type", "") + + +def strip_html(text: str | None) -> str: + if not text: + return "" + return _WS_RE.sub(" ", html.unescape(_TAG_RE.sub(" ", text))).strip() + + +def parse_date(value: str | None) -> str | None: + """Normalise RFC 822 (RSS) and ISO 8601 (Atom/JSON Feed) dates to UTC ISO.""" + if not value: + return None + value = value.strip() + try: + dt = parsedate_to_datetime(value) + except (TypeError, ValueError): + try: + dt = datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError: + return value + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + return dt.astimezone(timezone.utc).isoformat() + + +def _text(el, *paths) -> str | None: + for p in paths: + found = el.find(p, NS) + if found is not None and (found.text or "").strip(): + return found.text + return None + + +def _atom_link(entry) -> str | None: + alternate = None + for link in entry.findall("atom:link", NS) + entry.findall("link"): + href = link.get("href") + if not href: + continue + rel = link.get("rel", "alternate") + if rel == "alternate": + return href + alternate = alternate or href + return alternate + + +def parse_xml(data: bytes) -> dict: + root = ET.fromstring(data) + tag = root.tag.rsplit("}", 1)[-1].lower() + if tag == "feed": # Atom + title = strip_html(_text(root, "atom:title")) + entries = [] + for e in root.findall("atom:entry", NS): + entries.append({ + "title": strip_html(_text(e, "atom:title")), + "link": _atom_link(e), + "published": parse_date(_text(e, "atom:published", "atom:updated")), + "author": strip_html(_text(e, "atom:author/atom:name", "dc:creator")), + "summary": strip_html(_text(e, "atom:summary", "atom:content"))[:2000], + }) + return {"format": "atom", "title": title, "entries": entries} + channel = root.find("channel") if tag == "rss" else root # RSS 2.0 vs RDF/RSS 1.0 + if channel is None: + raise ValueError(f"unrecognised XML root <{tag}>") + entries = [] + for item in channel.iter("item") if tag == "rss" else root.iter("{http://purl.org/rss/1.0/}item"): + entries.append({ + "title": strip_html(_text(item, "title", "{http://purl.org/rss/1.0/}title")), + "link": (_text(item, "link", "{http://purl.org/rss/1.0/}link") or "").strip() or None, + "published": parse_date(_text(item, "pubDate", "dc:date")), + "author": strip_html(_text(item, "dc:creator", "author")), + "summary": strip_html(_text(item, "content:encoded", "description", "{http://purl.org/rss/1.0/}description"))[:2000], + }) + return {"format": "rss", "title": strip_html(_text(channel, "title", "{http://purl.org/rss/1.0/}title")), "entries": entries} + + +def parse_json_feed(data: bytes) -> dict: + doc = json.loads(data) + entries = [] + for item in doc.get("items", []): + authors = item.get("authors") or ([item["author"]] if item.get("author") else []) + entries.append({ + "title": strip_html(item.get("title")), + "link": item.get("url") or item.get("external_url"), + "published": parse_date(item.get("date_published") or item.get("date_modified")), + "author": ", ".join(a.get("name", "") for a in authors if isinstance(a, dict)) or None, + "summary": strip_html(item.get("summary") or item.get("content_text") or item.get("content_html"))[:2000], + }) + return {"format": "jsonfeed", "title": strip_html(doc.get("title")), "entries": entries} + + +def parse_feed(data: bytes, content_type: str = "") -> dict: + head = data.lstrip()[:1] + if head == b"{" or "json" in content_type: + return parse_json_feed(data) + return parse_xml(data) + + +def discover(page_url: str, page_html: bytes | None = None) -> list[str]: + """Return candidate feed URLs for a page: first, then well-known paths.""" + if page_html is None: + page_html, _ = fetch(page_url) + text = page_html.decode("utf-8", "replace") + found: list[str] = [] + for m in re.finditer(r"]*>", text, re.I): + tag = m.group(0) + type_m = re.search(r"""type\s*=\s*["']([^"']+)""", tag, re.I) + href_m = re.search(r"""href\s*=\s*["']([^"']+)""", tag, re.I) + rel_m = re.search(r"""rel\s*=\s*["']([^"']+)""", tag, re.I) + if not href_m or not type_m or type_m.group(1).lower() not in FEED_TYPES: + continue + if rel_m and "alternate" not in rel_m.group(1).lower(): + continue + url = urllib.parse.urljoin(page_url, html.unescape(href_m.group(1))) + if url not in found: + found.append(url) + if found: + return found + parsed = urllib.parse.urlsplit(page_url) + base = f"{parsed.scheme}://{parsed.netloc}" + return [base + p for p in COMMON_FEED_PATHS] + + +def looks_like_feed(data: bytes, content_type: str) -> bool: + head = data.lstrip()[:300].lower() + return head.startswith(b"{") and b"items" in data[:2000] or b" dict: + data, ctype = fetch(url) + if looks_like_feed(data, ctype): + feed = parse_feed(data, ctype) + feed["url"] = url + return feed + candidates = discover(url, data) + errors = [] + for cand in candidates: + try: + cdata, cctype = fetch(cand) + except (urllib.error.URLError, OSError) as exc: + errors.append(f"{cand}: {exc}") + continue + if looks_like_feed(cdata, cctype): + feed = parse_feed(cdata, cctype) + feed["url"] = cand + feed["discovered_from"] = url + return feed + raise SystemExit(f"no feed found at {url}; tried {len(candidates)} candidates\n" + "\n".join(errors)) + + +def filter_entries(entries: list[dict], limit: int, since: str | None) -> list[dict]: + if since: + cutoff = datetime.fromisoformat(since).replace(tzinfo=timezone.utc) if "T" not in since else datetime.fromisoformat(since.replace("Z", "+00:00")) + if cutoff.tzinfo is None: + cutoff = cutoff.replace(tzinfo=timezone.utc) + entries = [e for e in entries if e["published"] and datetime.fromisoformat(e["published"]) >= cutoff] + entries.sort(key=lambda e: e["published"] or "", reverse=True) + return entries[:limit] + + +def render_text(feed: dict) -> str: + lines = [f"{feed.get('title') or '(untitled feed)'} [{feed['format']}] {feed['url']}"] + for e in feed["entries"]: + when = (e["published"] or "")[:10] + by = f" — {e['author']}" if e.get("author") else "" + lines.append(f"- {when} {e['title'] or '(no title)'}{by}\n {e['link'] or ''}") + if e.get("summary"): + lines.append(f" {e['summary'][:300]}") + return "\n".join(lines) + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + sub = ap.add_subparsers(dest="cmd", required=True) + r = sub.add_parser("read", help="read a feed (or discover one behind a page URL)") + r.add_argument("url") + r.add_argument("--limit", type=int, default=20) + r.add_argument("--since", help="ISO date/datetime; drop older entries") + r.add_argument("--json", action="store_true") + d = sub.add_parser("discover", help="list feed URLs advertised by a page") + d.add_argument("url") + d.add_argument("--json", action="store_true") + args = ap.parse_args(argv) + + try: + if args.cmd == "discover": + urls = discover(args.url) + print(json.dumps(urls, indent=2) if args.json else "\n".join(urls)) + return 0 + feed = read(args.url) + feed["entries"] = filter_entries(feed["entries"], args.limit, args.since) + print(json.dumps(feed, indent=2, ensure_ascii=False) if args.json else render_text(feed)) + return 0 + except urllib.error.HTTPError as exc: + print(f"HTTP {exc.code} for {exc.url}", file=sys.stderr) + return 2 + except (urllib.error.URLError, ET.ParseError, ValueError, json.JSONDecodeError) as exc: + print(f"error: {exc}", file=sys.stderr) + return 2 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/social-media/reddit-reading/SKILL.md b/skills/social-media/reddit-reading/SKILL.md new file mode 100644 index 0000000000..8d97c6ece2 --- /dev/null +++ b/skills/social-media/reddit-reading/SKILL.md @@ -0,0 +1,107 @@ +--- +name: reddit-reading +description: "Read Reddit: subreddits, search, threads, users. No browser." +version: 1.0.0 +author: Teknium (teknium1), Hermes Agent +license: MIT +platforms: [linux, macos, windows] +metadata: + hermes: + tags: [Reddit, Social Media, Research, Discussions, Community] + related_skills: [rss-feeds, grounded-citations, blocked-page-recovery, xurl] +--- + +# Reddit Reading Skill + +Reads Reddit content — subreddit listings, site or subreddit search, full threads with +comments, and user activity — from a server or headless machine where the normal routes +are dead. It does not post, vote, or log in as a user. Idea credit: the per-platform +backend routing in [Agent Reach](https://github.com/Panniantong/Agent-Reach). + +## When to Use + +- "What is r/LocalLLaMA saying about X", "find Reddit threads on Y", "summarise this + Reddit thread", "what has u/someone posted lately". +- Any `reddit.com` URL the user shares. `web_extract`, `browser_navigate` and the + `.json` endpoints all fail from server IPs (403 or a "Prove your humanity" wall); + this skill is the working path. +- Not for posting, voting, messaging, or anything needing a user login. + +## Prerequisites + +None for the anonymous path. For anything beyond a handful of calls per task, create +a free Reddit "script" app at https://www.reddit.com/prefs/apps and put the two values +in `~/.hermes/.env`: + +``` +REDDIT_CLIENT_ID=... +REDDIT_CLIENT_SECRET=... +``` + +The script picks the OAuth backend automatically when both are set (~100 requests per +minute, scores, comment nesting, `num_comments`). Without them it uses Reddit's Atom +feeds, which are the only unauthenticated endpoints still served to non-residential IPs. + +## How to Run + +Run every command through `terminal` with the skill-relative script path: + +```bash +python3 scripts/reddit.py doctor # which backend, current rate-limit window +python3 scripts/reddit.py sub LocalLLaMA --sort hot --limit 15 +python3 scripts/reddit.py search "hermes agent" --sub LocalLLaMA --sort new +python3 scripts/reddit.py thread https://www.reddit.com/r/x/comments/abc123/slug/ --limit 40 +python3 scripts/reddit.py user spez --limit 10 +python3 scripts/reddit.py --json search "topic" # machine-readable +``` + +## Quick Reference + +| Need | Command | Anonymous | OAuth | +|---|---|---|---| +| Subreddit front page | `sub NAME --sort hot\|new\|top\|rising [--time week]` | ✔ | ✔ | +| Search all of Reddit | `search "q" --sort relevance\|new\|top\|comments` | ✔ | ✔ | +| Search one subreddit | `search "q" --sub NAME` | ✔ | ✔ | +| Thread + comments | `thread URL --limit N` | ✔ top-level only, no scores | ✔ nested, scores | +| User posts/comments | `user NAME` | ✔ | ✔ | +| Backend + rate limit | `doctor` | ✔ | ✔ | + +## Procedure + +① `doctor` once per task if you have not called it this session — it tells you which +backend is live and how many seconds remain in the anonymous window. + +② Plan your calls before making them. Anonymous Reddit allows roughly **one request per +minute per IP**; the script sleeps until the window resets on a 429 and retries once, so +a five-call plan costs about five minutes. Prefer one `search --sub` over several `sub` +listings, and read one thread rather than the whole listing. + +③ For "what is the community saying" questions, read the thread bodies (`thread`) rather +than stopping at titles; the listing only carries the first ~300 characters of each post. + +④ Cite the permalink (`url` field), not the listing page, when the result feeds a report. +`grounded-citations` registers these URLs like any other source. + +⑤ If the user needs sustained Reddit access (monitoring, more than ~10 calls), stop and +ask them to add the OAuth credentials rather than grinding through the throttle. + +## Pitfalls + +- `www.reddit.com/…/.json`, `api.reddit.com` and `old.reddit.com` return 403 or an + empty "Welcome to Reddit" shell for datacentre IPs. Do not fall back to them; do not + spoof a browser User-Agent (also 403). +- `r.jina.ai` and the `browser_navigate` tool hit the same block ("blocked by network + security" / humanity check). `blocked-page-recovery`'s Wayback route can still recover + an **old** thread that was archived; it cannot fetch fresh ones. +- Anonymous thread feeds only contain the post plus top-level comments (Reddit caps the + feed at a handful of entries); scores and reply nesting are OAuth-only. +- Reddit's `limit` on feeds is advisory — expect 5–25 entries regardless of what you ask. +- Never paste `REDDIT_CLIENT_SECRET` into a chat or log; the script reads it from the + environment only. + +## Verification + +`python3 scripts/reddit.py doctor` prints `anonymous_feed: ok` and an +`x-ratelimit-reset` value; `sub announcements --limit 1` returns one entry with a +`reddit.com/r/announcements/comments/` URL. With credentials set, `doctor` prints +`active_backend: oauth` and `thread …` output shows numeric scores. diff --git a/skills/social-media/reddit-reading/scripts/reddit.py b/skills/social-media/reddit-reading/scripts/reddit.py new file mode 100644 index 0000000000..3f00646437 --- /dev/null +++ b/skills/social-media/reddit-reading/scripts/reddit.py @@ -0,0 +1,307 @@ +#!/usr/bin/env python3 +"""Read Reddit without a browser: listings, search, threads with comments, user pages. + +Two backends, chosen automatically: + +* **OAuth API** (preferred when ``REDDIT_CLIENT_ID`` + ``REDDIT_CLIENT_SECRET`` are set): + a free "script" app from https://www.reddit.com/prefs/apps; ~100 requests/minute, + full JSON including scores and nested comments. +* **Anonymous Atom feeds** (``.rss`` endpoints): the only unauthenticated path Reddit + still serves to server IPs (``.json`` and old.reddit return 403 / an empty shell). + Roughly ONE request per minute per IP; the script sleeps until the window resets + when it hits a 429 and retries once. + + python3 reddit.py sub LocalLLaMA [--sort hot|new|top] [--limit N] + python3 reddit.py search "hermes agent" [--sub LocalLLaMA] [--sort new] [--limit N] + python3 reddit.py thread https://www.reddit.com/r/x/comments/abc123/... [--limit N] + python3 reddit.py user spez [--limit N] + python3 reddit.py doctor # which backend is active, and why + +Add ``--json`` to any read command for machine-readable output. Standard library only. +""" + +from __future__ import annotations + +import argparse +import base64 +import html +import json +import os +import re +import sys +import time +import urllib.error +import urllib.parse +import urllib.request +import xml.etree.ElementTree as ET + +USER_AGENT = "hermes-agent/1.0 (reddit-reading skill; +https://github.com/NousResearch/hermes-agent)" +TIMEOUT = 25 +ATOM = {"a": "http://www.w3.org/2005/Atom"} +WWW = "https://www.reddit.com" +OAUTH = "https://oauth.reddit.com" +_TAG_RE = re.compile(r"<[^>]+>") +_WS_RE = re.compile(r"\s+") +_THREAD_RE = re.compile(r"reddit\.com/r/([^/]+)/comments/([a-z0-9]+)", re.I) + + +def strip_html(text: str | None) -> str: + if not text: + return "" + # Reddit wraps entry bodies in a with a "submitted by /u/x [link] [comments]" footer. + text = _TAG_RE.sub(" ", html.unescape(text)) + text = re.sub(r"submitted by\s+/u/\S+|\[link\]|\[comments\]", " ", text) + return _WS_RE.sub(" ", html.unescape(text)).strip() + + +# ── HTTP ───────────────────────────────────────────────────────────────────── + +def _get(url: str, headers: dict | None = None, retry_on_429: bool = True) -> tuple[bytes, dict]: + hdrs = {"User-Agent": USER_AGENT, "Accept": "*/*"} + hdrs.update(headers or {}) + req = urllib.request.Request(url, headers=hdrs) + try: + with urllib.request.urlopen(req, timeout=TIMEOUT) as resp: + return resp.read(), dict(resp.headers) + except urllib.error.HTTPError as exc: + if exc.code == 429 and retry_on_429: + wait = _reset_seconds(exc.headers) + print(f"reddit: 429 rate-limited, sleeping {wait}s until the window resets", file=sys.stderr) + time.sleep(wait) + return _get(url, headers, retry_on_429=False) + raise + + +def _reset_seconds(headers) -> int: + for key in ("x-ratelimit-reset", "retry-after"): + val = headers.get(key) if headers else None + if val: + try: + return max(1, min(int(float(val)) + 1, 120)) + except ValueError: + pass + return 61 + + +# ── OAuth backend ──────────────────────────────────────────────────────────── + +def oauth_credentials() -> tuple[str, str] | None: + cid, secret = os.environ.get("REDDIT_CLIENT_ID"), os.environ.get("REDDIT_CLIENT_SECRET") + return (cid, secret) if cid and secret else None + + +def oauth_token(cid: str, secret: str) -> str: + body = urllib.parse.urlencode({"grant_type": "client_credentials"}).encode() + auth = base64.b64encode(f"{cid}:{secret}".encode()).decode() + req = urllib.request.Request( + f"{WWW}/api/v1/access_token", data=body, + headers={"Authorization": f"Basic {auth}", "User-Agent": USER_AGENT}, + ) + with urllib.request.urlopen(req, timeout=TIMEOUT) as resp: + return json.loads(resp.read())["access_token"] + + +def _api(path: str, token: str, **params): + params.setdefault("raw_json", 1) + url = f"{OAUTH}{path}?{urllib.parse.urlencode({k: v for k, v in params.items() if v is not None})}" + data, _ = _get(url, {"Authorization": f"Bearer {token}"}) + return json.loads(data) + + +def _post_from_api(child: dict) -> dict: + d = child["data"] + return { + "title": d.get("title"), + "author": d.get("author"), + "subreddit": d.get("subreddit"), + "score": d.get("score"), + "num_comments": d.get("num_comments"), + "created_utc": d.get("created_utc"), + "url": f"{WWW}{d['permalink']}" if d.get("permalink") else d.get("url"), + "external_url": None if d.get("is_self") else d.get("url"), + "body": (d.get("selftext") or "")[:4000], + } + + +def _flatten_comments(children: list, depth: int = 0, out: list | None = None) -> list: + out = out if out is not None else [] + for c in children: + if c.get("kind") != "t1": + continue + d = c["data"] + out.append({ + "author": d.get("author"), "score": d.get("score"), "depth": depth, + "created_utc": d.get("created_utc"), "body": (d.get("body") or "")[:4000], + "url": f"{WWW}{d['permalink']}" if d.get("permalink") else None, + }) + replies = d.get("replies") + if isinstance(replies, dict): + _flatten_comments(replies["data"]["children"], depth + 1, out) + return out + + +def api_listing(token: str, path: str, limit: int, **params) -> list[dict]: + data = _api(path, token, limit=limit, **params) + return [_post_from_api(c) for c in data["data"]["children"] if c.get("kind") == "t3"] + + +def api_thread(token: str, sub: str, post_id: str, limit: int) -> dict: + data = _api(f"/r/{sub}/comments/{post_id}", token, limit=limit, depth=10, sort="top") + post = _post_from_api(data[0]["data"]["children"][0]) + post["comments"] = _flatten_comments(data[1]["data"]["children"])[:limit] + return post + + +# ── Anonymous Atom backend ─────────────────────────────────────────────────── + +def _entries(url: str) -> list[dict]: + data, _ = _get(url) + root = ET.fromstring(data) + out = [] + for e in root.findall("a:entry", ATOM): + link = e.find("a:link", ATOM) + out.append({ + "title": strip_html(e.findtext("a:title", default="", namespaces=ATOM)), + "author": (e.findtext("a:author/a:name", default="", namespaces=ATOM) or "").replace("/u/", "") or None, + "created": e.findtext("a:updated", default="", namespaces=ATOM) or None, + "url": link.get("href") if link is not None else None, + "body": strip_html(e.findtext("a:content", default="", namespaces=ATOM))[:4000], + }) + return out + + +def atom_listing(path: str, limit: int, **params) -> list[dict]: + params["limit"] = limit + return _entries(f"{WWW}{path}.rss?{urllib.parse.urlencode({k: v for k, v in params.items() if v is not None})}") + + +def atom_thread(sub: str, post_id: str, limit: int) -> dict: + entries = _entries(f"{WWW}/r/{sub}/comments/{post_id}/.rss?limit={limit}") + if not entries: + raise SystemExit("thread feed returned no entries") + post, comments = entries[0], entries[1:] + post["comments"] = [{"author": c["author"], "created": c["created"], "body": c["body"], "url": c["url"]} for c in comments] + post["note"] = "anonymous feed: scores and nesting unavailable; set REDDIT_CLIENT_ID/SECRET for full data" + return post + + +# ── Commands ───────────────────────────────────────────────────────────────── + +def parse_thread_url(url: str) -> tuple[str, str]: + m = _THREAD_RE.search(url) + if not m: + raise SystemExit(f"not a Reddit thread URL: {url}") + return m.group(1), m.group(2) + + +def cmd_sub(a, token): + path = f"/r/{a.name}/{a.sort}" + if token: + return api_listing(token, path, a.limit, t=a.time if a.sort == "top" else None) + return atom_listing(path, a.limit, t=a.time if a.sort == "top" else None) + + +def cmd_search(a, token): + path = f"/r/{a.sub}/search" if a.sub else "/search" + params = {"q": a.query, "sort": a.sort, "restrict_sr": 1 if a.sub else None, "t": a.time} + return api_listing(token, path, a.limit, **params) if token else atom_listing(path, a.limit, **params) + + +def cmd_thread(a, token): + sub, post_id = parse_thread_url(a.url) + return api_thread(token, sub, post_id, a.limit) if token else atom_thread(sub, post_id, a.limit) + + +def cmd_user(a, token): + path = f"/user/{a.name}" + if token: + data = _api(f"{path}/overview", token, limit=a.limit) + out = [] + for c in data["data"]["children"]: + out.append(_post_from_api(c) if c["kind"] == "t3" else _flatten_comments([c])[0]) + return out + return atom_listing(path, a.limit) + + +def cmd_doctor(a, token): + report = {"oauth_credentials": bool(oauth_credentials()), "user_agent": USER_AGENT} + if token: + try: + _api("/r/announcements/hot", token, limit=1) + report["active_backend"] = "oauth" + except (urllib.error.URLError, OSError, KeyError) as exc: + report["active_backend"] = "oauth (broken)" + report["oauth_error"] = str(exc) + else: + report["active_backend"] = "anonymous-atom" + try: + data, headers = _get(f"{WWW}/r/announcements/.rss?limit=1", retry_on_429=False) + report["anonymous_feed"] = "ok" if b" str: + if cmd == "doctor": + return "\n".join(f"{k}: {v}" for k, v in result.items()) + if cmd == "thread": + p = result + lines = [f"# {p.get('title')} — u/{p.get('author')} score={p.get('score', '?')} {p.get('url')}", p.get("body", "")[:1500], ""] + for c in p["comments"]: + indent = " " * c.get("depth", 0) + lines.append(f"{indent}- u/{c.get('author')} (score {c.get('score', '?')}): {c.get('body', '')[:600]}") + if p.get("note"): + lines.append(f"\n[{p['note']}]") + return "\n".join(lines) + lines = [] + for p in result: + score = f" ↑{p['score']}" if p.get("score") is not None else "" + nc = f" 💬{p['num_comments']}" if p.get("num_comments") is not None else "" + lines.append(f"- {p.get('title') or p.get('body', '')[:80]}{score}{nc} — u/{p.get('author')}\n {p.get('url')}") + if p.get("body") and p.get("title"): + lines.append(f" {p['body'][:300]}") + return "\n".join(lines) or "(no results)" + + +def main(argv: list[str] | None = None) -> int: + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--json", action="store_true") + sub = ap.add_subparsers(dest="cmd", required=True) + s = sub.add_parser("sub"); s.add_argument("name"); s.add_argument("--sort", default="hot", choices=["hot", "new", "top", "rising"]); s.add_argument("--time", default="week", choices=["hour", "day", "week", "month", "year", "all"]); s.add_argument("--limit", type=int, default=15) + q = sub.add_parser("search"); q.add_argument("query"); q.add_argument("--sub"); q.add_argument("--sort", default="relevance", choices=["relevance", "new", "top", "comments"]); q.add_argument("--time", default="all", choices=["hour", "day", "week", "month", "year", "all"]); q.add_argument("--limit", type=int, default=15) + t = sub.add_parser("thread"); t.add_argument("url"); t.add_argument("--limit", type=int, default=40) + u = sub.add_parser("user"); u.add_argument("name"); u.add_argument("--limit", type=int, default=15) + sub.add_parser("doctor") + args = ap.parse_args(argv) + + creds = oauth_credentials() + token = None + if creds: + try: + token = oauth_token(*creds) + except (urllib.error.URLError, OSError, KeyError) as exc: + print(f"reddit: OAuth token failed ({exc}); falling back to anonymous feeds", file=sys.stderr) + try: + result = COMMANDS[args.cmd](args, token) + except urllib.error.HTTPError as exc: + print(f"HTTP {exc.code} for {exc.url}", file=sys.stderr) + return 2 + except (urllib.error.URLError, ET.ParseError, json.JSONDecodeError, KeyError) as exc: + print(f"error: {exc}", file=sys.stderr) + return 2 + print(json.dumps(result, indent=2, ensure_ascii=False) if args.json else render(args.cmd, result)) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/skills/test_reddit_reading_skill.py b/tests/skills/test_reddit_reading_skill.py new file mode 100644 index 0000000000..45310f098d --- /dev/null +++ b/tests/skills/test_reddit_reading_skill.py @@ -0,0 +1,113 @@ +"""Tests for skills/social-media/reddit-reading/scripts/reddit.py — backend selection and throttle handling.""" + +import io +import sys +import urllib.error +from pathlib import Path +from unittest import mock + +import pytest + +SCRIPTS_DIR = Path(__file__).resolve().parents[2] / "skills" / "social-media" / "reddit-reading" / "scripts" +sys.path.insert(0, str(SCRIPTS_DIR)) + +import reddit # noqa: E402 + +THREAD_ATOM = b""" +/u/opPost title + 2026-09-01T00:00:00+00:00 + <div>body text</div> submitted by /u/op [link] [comments] +/u/c1/u/c1 on Post title + 2026-09-01T01:00:00+00:00 + <p>first comment</p> +""" + + +def _http_error(code, headers): + return urllib.error.HTTPError("https://www.reddit.com/x", code, "msg", headers, io.BytesIO(b"")) + + +def test_anonymous_thread_uses_atom_feed_and_waits_out_a_429_exactly_once(monkeypatch): + """Without OAuth credentials the .rss endpoint is used; a 429 sleeps for x-ratelimit-reset and retries once, + and the parsed thread separates the post from its comments with feed noise stripped.""" + monkeypatch.delenv("REDDIT_CLIENT_ID", raising=False) + monkeypatch.delenv("REDDIT_CLIENT_SECRET", raising=False) + assert reddit.oauth_credentials() is None + + calls = [] + sleeps = [] + + class Resp(io.BytesIO): + headers = {"x-ratelimit-remaining": "0.0"} + + def __enter__(self): + return self + + def __exit__(self, *a): + self.close() + + def fake_urlopen(req, timeout): + calls.append(req.full_url) + if len(calls) == 1: + raise _http_error(429, {"x-ratelimit-reset": "7"}) + return Resp(THREAD_ATOM) + + with mock.patch.object(reddit.urllib.request, "urlopen", fake_urlopen), \ + mock.patch.object(reddit.time, "sleep", sleeps.append): + post = reddit.atom_thread("test", "abc123", limit=10) + + assert calls[0].startswith("https://www.reddit.com/r/test/comments/abc123/.rss") and len(calls) == 2 + assert sleeps == [8] # reset + 1s margin, one retry only + assert post["title"] == "Post title" and post["author"] == "op" + assert post["body"] == "body text" # "submitted by … [link] [comments]" footer stripped + assert [c["author"] for c in post["comments"]] == ["c1"] + + # a second 429 after the retry propagates instead of looping + with mock.patch.object(reddit.urllib.request, "urlopen", side_effect=_http_error(429, {})), \ + mock.patch.object(reddit.time, "sleep", lambda s: None), pytest.raises(urllib.error.HTTPError): + reddit._get("https://www.reddit.com/r/test/.rss") + + +def test_oauth_credentials_route_to_oauth_host_and_flatten_nested_comments(monkeypatch): + """With REDDIT_CLIENT_ID/SECRET the script talks to oauth.reddit.com with a bearer token and + returns nested comments flattened with depth and scores — data the anonymous path cannot provide.""" + monkeypatch.setenv("REDDIT_CLIENT_ID", "cid") + monkeypatch.setenv("REDDIT_CLIENT_SECRET", "sec") + assert reddit.oauth_credentials() == ("cid", "sec") + + listing = [ + {"data": {"children": [{"kind": "t3", "data": {"title": "T", "author": "op", "subreddit": "test", "score": 42, + "num_comments": 2, "created_utc": 1.0, "permalink": "/r/test/comments/abc123/t/", + "is_self": True, "url": "https://www.reddit.com/r/test/comments/abc123/t/", "selftext": "s"}}]}}, + {"data": {"children": [{"kind": "t1", "data": {"author": "a", "score": 5, "body": "top", "permalink": "/p/1", + "replies": {"data": {"children": [{"kind": "t1", "data": {"author": "b", "score": 1, "body": "reply", "replies": ""}}]}}}}, + {"kind": "more", "data": {}}]}}, + ] + seen = {} + + def fake_api(path, token, **params): + seen["path"], seen["token"] = path, token + return listing + + with mock.patch.object(reddit, "_api", fake_api): + post = reddit.api_thread("tok", "test", "abc123", limit=10) + + assert seen == {"path": "/r/test/comments/abc123", "token": "tok"} + assert post["score"] == 42 and post["url"] == "https://www.reddit.com/r/test/comments/abc123/t/" + assert [(c["author"], c["depth"], c["score"]) for c in post["comments"]] == [("a", 0, 5), ("b", 1, 1)] + + # the bearer header actually reaches the OAuth host + captured = {} + + def fake_get(url, headers=None, retry_on_429=True): + captured["url"], captured["headers"] = url, headers + return b'{"data": {"children": []}}', {} + + with mock.patch.object(reddit, "_get", fake_get): + reddit._api("/r/test/hot", "tok", limit=1) + assert captured["url"].startswith("https://oauth.reddit.com/r/test/hot?") + assert captured["headers"]["Authorization"] == "Bearer tok" + + +if __name__ == "__main__": + sys.exit(pytest.main([__file__])) diff --git a/tests/skills/test_rss_feeds_skill.py b/tests/skills/test_rss_feeds_skill.py new file mode 100644 index 0000000000..9ce7af8748 --- /dev/null +++ b/tests/skills/test_rss_feeds_skill.py @@ -0,0 +1,69 @@ +"""Tests for skills/research/rss-feeds/scripts/feed.py — parsing and discovery contracts.""" + +import sys +from pathlib import Path +from unittest import mock + +import pytest + +SCRIPTS_DIR = Path(__file__).resolve().parents[2] / "skills" / "research" / "rss-feeds" / "scripts" +sys.path.insert(0, str(SCRIPTS_DIR)) + +import feed # noqa: E402 + +RSS = b""" +Blog & Notes +Olderhttps://ex.com/aMon, 01 Sep 2026 10:00:00 GMT + Ann<p>Hello <b>world</b></p> +Newerhttps://ex.com/bThu, 04 Sep 2026 08:30:00 +0200 +""" + +ATOM = b"""Atom Site +Entry +2026-09-03T12:00:00ZBob<p>Body</p> +""" + +JSONFEED = b'{"version":"https://jsonfeed.org/version/1.1","title":"JF","items":[{"id":"1","url":"https://ex.com/j","title":"J1","date_published":"2026-09-02T00:00:00Z","authors":[{"name":"Cy"}],"content_text":"txt"}]}' + +PAGE = b"""x + +""" + + +def test_all_three_formats_normalise_to_the_same_entry_shape_and_utc_dates(): + """RSS (RFC 822), Atom (ISO), JSON Feed (ISO) parse to identical keys with UTC-normalised dates, + and Atom picks the rel=alternate link over rel=self.""" + rss, atom, jf = feed.parse_feed(RSS), feed.parse_feed(ATOM), feed.parse_feed(JSONFEED, "application/feed+json") + keys = {"title", "link", "published", "author", "summary"} + for f in (rss, atom, jf): + assert f["entries"] and all(set(e) == keys for e in f["entries"]) + assert rss["title"] == "Blog & Notes" + assert rss["entries"][0]["summary"] == "Hello world" # HTML stripped, entities decoded + assert rss["entries"][1]["published"] == "2026-09-04T06:30:00+00:00" # +0200 → UTC + assert atom["entries"][0]["link"] == "https://ex.com/post" + assert jf["entries"][0]["author"] == "Cy" + newest = feed.filter_entries(rss["entries"], limit=1, since="2026-09-02") + assert [e["title"] for e in newest] == ["Newer"] + + +def test_page_url_discovers_advertised_feed_then_reads_it(): + """A non-feed page falls through to discovery (resolved against the page URL) + and `read` returns the parsed feed tagged with where it was discovered from.""" + responses = { + "https://ex.com/blog/": (PAGE, "text/html"), + "https://ex.com/atom/everything/": (ATOM, "application/atom+xml"), + } + with mock.patch.object(feed, "fetch", side_effect=lambda u: responses[u]): + assert feed.discover("https://ex.com/blog/") == ["https://ex.com/atom/everything/"] + result = feed.read("https://ex.com/blog/") + assert result["url"] == "https://ex.com/atom/everything/" + assert result["discovered_from"] == "https://ex.com/blog/" + assert result["entries"][0]["title"] == "Entry" + # no advertised feed → well-known paths are proposed, never an empty list + with mock.patch.object(feed, "fetch", return_value=(b"plain", "text/html")): + candidates = feed.discover("https://plain.example/") + assert candidates and all(c.startswith("https://plain.example/") for c in candidates) + + +if __name__ == "__main__": + sys.exit(pytest.main([__file__])) diff --git a/website/docs/reference/skills-catalog.md b/website/docs/reference/skills-catalog.md index b6ab7f5c21..d56a104fa9 100644 --- a/website/docs/reference/skills-catalog.md +++ b/website/docs/reference/skills-catalog.md @@ -101,11 +101,13 @@ If a skill is missing from this list but present in the repo, the catalog is reg | [`competitor-news-monitor`](/docs/user-guide/skills/bundled/research/research-competitor-news-monitor) | Watch named companies for material news; cited digests. | `research\competitor-news-monitor` | | [`grounded-citations`](/docs/user-guide/skills/bundled/research/research-grounded-citations) | Ground answers and documents in cited, verifiable sources. | `research\grounded-citations` | | [`llm-wiki`](/docs/user-guide/skills/bundled/research/research-llm-wiki) | Karpathy's LLM Wiki: build/query interlinked markdown KB. | `research\llm-wiki` | +| [`rss-feeds`](/docs/user-guide/skills/bundled/research/research-rss-feeds) | Read RSS, Atom, JSON feeds; discover feeds behind a page. | `research/rss-feeds` | ## social-media | Skill | Description | Path | |-------|-------------|------| +| [`reddit-reading`](/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading) | Read Reddit: subreddits, search, threads, users. No browser. | `social-media/reddit-reading` | | [`xurl`](/docs/user-guide/skills/bundled/social-media/social-media-xurl) | X/Twitter via xurl CLI: raw post search, posting, DM, media. | `social-media\xurl` | ## software-development diff --git a/website/docs/user-guide/skills/bundled/research/research-competitor-news-monitor.md b/website/docs/user-guide/skills/bundled/research/research-competitor-news-monitor.md index dd756be8a9..5ac27e64d4 100644 --- a/website/docs/user-guide/skills/bundled/research/research-competitor-news-monitor.md +++ b/website/docs/user-guide/skills/bundled/research/research-competitor-news-monitor.md @@ -15,13 +15,13 @@ Watch named companies for material news; cited digests. | | | |---|---| | Source | Bundled (installed by default) | -| Path | `skills/research\competitor-news-monitor` | +| Path | `skills/research/competitor-news-monitor` | | Version | `0.1.0` | | Author | Ben Barclay (benbarclay), Hermes Agent | | License | MIT | | Platforms | linux, macos, windows | | Tags | `Competitors`, `News`, `Market-Research`, `Monitoring` | -| Related skills | [`blogwatcher`](/docs/user-guide/skills/optional/research/research-blogwatcher) | +| Related skills | [`blogwatcher`](/docs/user-guide/skills/optional/research/research-blogwatcher), [`rss-feeds`](/docs/user-guide/skills/bundled/research/research-rss-feeds), [`reddit-reading`](/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading) | ## Reference: full SKILL.md @@ -60,7 +60,7 @@ For each company include, where available: 5. reputable trade and financial press 6. job postings as weak supporting evidence -Use `blogwatcher` for feeds and `web_search`/`web_extract` for pages. Write the watch contract (watchlist, categories, materiality threshold, last cutoff) to a state file under `~/.hermes/competitor-watches/.json`, then create the job: +Use `rss-feeds` (bundled) or `blogwatcher` (optional, stateful) for feeds, `reddit-reading` for community discussion, and `web_search`/`web_extract` for pages. Write the watch contract (watchlist, categories, materiality threshold, last cutoff) to a state file under `~/.hermes/competitor-watches/.json`, then create the job: ``` cronjob(action="create", diff --git a/website/docs/user-guide/skills/bundled/research/research-grounded-citations.md b/website/docs/user-guide/skills/bundled/research/research-grounded-citations.md index 938e223a68..8166801512 100644 --- a/website/docs/user-guide/skills/bundled/research/research-grounded-citations.md +++ b/website/docs/user-guide/skills/bundled/research/research-grounded-citations.md @@ -15,13 +15,13 @@ Ground answers and documents in cited, verifiable sources. | | | |---|---| | Source | Bundled (installed by default) | -| Path | `skills/research\grounded-citations` | -| Version | `1.1.0` | +| Path | `skills/research/grounded-citations` | +| Version | `1.2.0` | | Author | Hermes Agent + Teknium | | License | MIT | | Platforms | linux, macos, windows | | Tags | `Research`, `Citations`, `Grounding`, `Sources`, `Web`, `Reports` | -| Related skills | [`arxiv`](/docs/user-guide/skills/bundled/research/research-arxiv), [`arxiv`](/docs/user-guide/skills/bundled/research/research-arxiv), `ocr-and-documents` | +| Related skills | [`arxiv`](/docs/user-guide/skills/bundled/research/research-arxiv), [`pdf`](/docs/user-guide/skills/bundled/productivity/productivity-pdf), [`reddit-reading`](/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading), [`rss-feeds`](/docs/user-guide/skills/bundled/research/research-rss-feeds), [`youtube-content`](/docs/user-guide/skills/bundled/media/media-youtube-content) | ## Reference: full SKILL.md @@ -143,6 +143,27 @@ sources, cite inline, end with the rendered `Sources:` list. For a short answer you may render the block from `sources.py render --only ` instead of writing to a file. +## Multi-Platform Sweeps + +"What are people saying about X" / "research X across the web" is not one +`web_search`. Fan out across source types, collect in parallel, then synthesise +with every claim attributed to the platform it came from: + +| Source type | Route | What it adds | +|---|---|---| +| Open web | `web_search` → `web_extract` | official docs, articles, announcements | +| Community discussion | `reddit-reading` (`search`, `thread`) | real user experience, complaints, workarounds | +| Blogs / releases / changelogs | `rss-feeds` (`read`, `discover`) | dated primary posts, version history | +| Video | `youtube-content` | walkthroughs, demos, talks | +| Code | `terminal` with `gh search repos` / `gh search issues` | implementations, open bugs | +| X/Twitter | `xurl` (needs API access) | announcements, developer chatter | + +Register every URL from every route in the ledger as it arrives (step ②). Keep +opinion and measurement apart: a Reddit thread is evidence that users *report* +something, not that it is true; pair it with a primary source or label it as +sentiment. Report per-platform coverage gaps ("Reddit search returned nothing +newer than March") rather than silently narrowing to what worked. + ## Fact-Checking Mode For work where the reader must be able to check the chain — medical, legal, diff --git a/website/docs/user-guide/skills/bundled/research/research-rss-feeds.md b/website/docs/user-guide/skills/bundled/research/research-rss-feeds.md new file mode 100644 index 0000000000..4946eaa94b --- /dev/null +++ b/website/docs/user-guide/skills/bundled/research/research-rss-feeds.md @@ -0,0 +1,114 @@ +--- +title: "Rss Feeds — Read RSS, Atom, JSON feeds; discover feeds behind a page" +sidebar_label: "Rss Feeds" +description: "Read RSS, Atom, JSON feeds; discover feeds behind a page" +--- + +{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */} + +# Rss Feeds + +Read RSS, Atom, JSON feeds; discover feeds behind a page. + +## Skill metadata + +| | | +|---|---| +| Source | Bundled (installed by default) | +| Path | `skills/research/rss-feeds` | +| Version | `1.0.0` | +| Author | Teknium (teknium1), Hermes Agent | +| License | MIT | +| Platforms | linux, macos, windows | +| Tags | `RSS`, `Atom`, `Feeds`, `Monitoring`, `Research`, `Blogs`, `Releases` | +| Related skills | [`reddit-reading`](/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading), [`competitor-news-monitor`](/docs/user-guide/skills/bundled/research/research-competitor-news-monitor), [`grounded-citations`](/docs/user-guide/skills/bundled/research/research-grounded-citations), [`youtube-content`](/docs/user-guide/skills/bundled/media/media-youtube-content), [`blogwatcher`](/docs/user-guide/skills/optional/research/research-blogwatcher) | + +## Reference: full SKILL.md + +:::info +The following is the complete skill definition that Hermes loads when this skill is triggered. This is what the agent sees as instructions when the skill is active. +::: + +# RSS Feeds Skill + +Reads any RSS 2.0, RSS 1.0/RDF, Atom, or JSON Feed URL into a clean, date-sorted list of +entries, and discovers the feed behind an ordinary page URL (`` or +the usual `/feed`, `/rss.xml`, `/atom.xml` paths). Standard library only, nothing to +install. It does not fetch full article bodies — pass an entry's link to `web_extract` for +that. + +## When to Use + +- "What's new on <blog/site>", "latest releases of <GitHub repo>", "recent posts in + <subreddit>", "read this feed", "does this site have an RSS feed". +- Building a recurring digest with `cronjob_manage` (feeds are cheaper and more stable than + scraping the HTML front page every run). For a persistent read/unread database across + many feeds install the optional `blogwatcher` skill; this skill is the zero-install read. +- Anything where a structured list of `title / link / date / author / summary` beats a + rendered page: podcasts, changelogs, YouTube channels, newsrooms, forum categories. + +## Prerequisites + +None. Python 3.10+, network access to the feed host. + +## How to Run + +Run through `terminal` with the skill-relative script path: + +```bash +python3 scripts/feed.py read https://hnrss.org/frontpage --limit 10 +python3 scripts/feed.py read https://simonwillison.net/ # page URL → discovers the feed +python3 scripts/feed.py read URL --since 2026-09-01 --json # only newer entries, machine-readable +python3 scripts/feed.py discover https://example.com/ # list candidate feed URLs +``` + +## Quick Reference + +| Source | Feed URL pattern | +|---|---| +| GitHub releases / commits / tags | `https://github.com/OWNER/REPO/releases.atom`, `…/commits/BRANCH.atom`, `…/tags.atom` | +| Subreddit / Reddit search | `https://www.reddit.com/r/NAME/.rss`, `https://www.reddit.com/search.rss?q=…` (1 req/min anon; see `reddit-reading`) | +| YouTube channel | `https://www.youtube.com/feeds/videos.xml?channel_id=UC…` | +| Hacker News | `https://hnrss.org/frontpage`, `https://hnrss.org/newest?q=TERM` | +| arXiv category | `https://rss.arxiv.org/rss/cs.CL` | +| Substack / Medium / WordPress / Ghost | `SITE/feed`, `medium.com/feed/@user`, `SITE/rss/` | +| Podcasts | the show's RSS URL from its hosting page (`discover` finds it) | + +Output fields per entry: `title`, `link`, `published` (UTC ISO 8601), `author`, `summary` +(HTML stripped, ≤ 2000 chars). Entries are sorted newest-first. + +## Procedure + +① If you only have a site URL, run `read` on it directly; the script discovers the feed +and reports which URL it used (`discovered_from`). Use `discover` when you want to choose +between several advertised feeds (comments feed vs posts feed, per-category feeds). + +② Bound the request: `--limit` for "latest N", `--since YYYY-MM-DD` for "since last +check". For a cron digest persist the last-seen `published` value and pass it as +`--since` next run. + +③ For full text, hand the entry `link` to `web_extract`; feed summaries are frequently +truncated or the first paragraph only. + +④ Cite the entry `link`, not the feed URL, when the result feeds a report +(`grounded-citations`). + +## Pitfalls + +- A 200 response with HTML means the URL is a page, not a feed; the script falls through + to discovery automatically, but a site with no `` and none of the + common paths reports `no feed found` — check the site's footer or `/sitemap.xml` before + concluding there is none. +- Reddit feeds share Reddit's anonymous throttle (about one request per minute per IP). + Chain them through `reddit-reading`, which waits out the window, when you need more + than one Reddit call. +- Dates: RSS `pubDate` is RFC 822 and Atom uses ISO 8601; the script normalises both to + UTC. Feeds that omit dates sort to the bottom and are dropped by `--since`. +- Some feeds are Cloudflare-fronted and 403 non-browser clients; `blocked-page-recovery` + handles that class. + +## Verification + +`python3 scripts/feed.py read https://github.com/NousResearch/hermes-agent/releases.atom +--limit 1` prints one entry with a `releases/tag/` link and a `[atom]` format tag; +`discover https://simonwillison.net/` prints an `/atom/` URL. diff --git a/website/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading.md b/website/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading.md new file mode 100644 index 0000000000..219195c947 --- /dev/null +++ b/website/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading.md @@ -0,0 +1,125 @@ +--- +title: "Reddit Reading — Read Reddit: subreddits, search, threads, users" +sidebar_label: "Reddit Reading" +description: "Read Reddit: subreddits, search, threads, users" +--- + +{/* This page is auto-generated from the skill's SKILL.md by website/scripts/generate-skill-docs.py. Edit the source SKILL.md, not this page. */} + +# Reddit Reading + +Read Reddit: subreddits, search, threads, users. No browser. + +## Skill metadata + +| | | +|---|---| +| Source | Bundled (installed by default) | +| Path | `skills/social-media/reddit-reading` | +| Version | `1.0.0` | +| Author | Teknium (teknium1), Hermes Agent | +| License | MIT | +| Platforms | linux, macos, windows | +| Tags | `Reddit`, `Social Media`, `Research`, `Discussions`, `Community` | +| Related skills | [`rss-feeds`](/docs/user-guide/skills/bundled/research/research-rss-feeds), [`grounded-citations`](/docs/user-guide/skills/bundled/research/research-grounded-citations), [`blocked-page-recovery`](/docs/user-guide/skills/bundled/web/web-blocked-page-recovery), [`xurl`](/docs/user-guide/skills/bundled/social-media/social-media-xurl) | + +## Reference: full SKILL.md + +:::info +The following is the complete skill definition that Hermes loads when this skill is triggered. This is what the agent sees as instructions when the skill is active. +::: + +# Reddit Reading Skill + +Reads Reddit content — subreddit listings, site or subreddit search, full threads with +comments, and user activity — from a server or headless machine where the normal routes +are dead. It does not post, vote, or log in as a user. Idea credit: the per-platform +backend routing in [Agent Reach](https://github.com/Panniantong/Agent-Reach). + +## When to Use + +- "What is r/LocalLLaMA saying about X", "find Reddit threads on Y", "summarise this + Reddit thread", "what has u/someone posted lately". +- Any `reddit.com` URL the user shares. `web_extract`, `browser_navigate` and the + `.json` endpoints all fail from server IPs (403 or a "Prove your humanity" wall); + this skill is the working path. +- Not for posting, voting, messaging, or anything needing a user login. + +## Prerequisites + +None for the anonymous path. For anything beyond a handful of calls per task, create +a free Reddit "script" app at https://www.reddit.com/prefs/apps and put the two values +in `~/.hermes/.env`: + +``` +REDDIT_CLIENT_ID=... +REDDIT_CLIENT_SECRET=... +``` + +The script picks the OAuth backend automatically when both are set (~100 requests per +minute, scores, comment nesting, `num_comments`). Without them it uses Reddit's Atom +feeds, which are the only unauthenticated endpoints still served to non-residential IPs. + +## How to Run + +Run every command through `terminal` with the skill-relative script path: + +```bash +python3 scripts/reddit.py doctor # which backend, current rate-limit window +python3 scripts/reddit.py sub LocalLLaMA --sort hot --limit 15 +python3 scripts/reddit.py search "hermes agent" --sub LocalLLaMA --sort new +python3 scripts/reddit.py thread https://www.reddit.com/r/x/comments/abc123/slug/ --limit 40 +python3 scripts/reddit.py user spez --limit 10 +python3 scripts/reddit.py --json search "topic" # machine-readable +``` + +## Quick Reference + +| Need | Command | Anonymous | OAuth | +|---|---|---|---| +| Subreddit front page | `sub NAME --sort hot\|new\|top\|rising [--time week]` | ✔ | ✔ | +| Search all of Reddit | `search "q" --sort relevance\|new\|top\|comments` | ✔ | ✔ | +| Search one subreddit | `search "q" --sub NAME` | ✔ | ✔ | +| Thread + comments | `thread URL --limit N` | ✔ top-level only, no scores | ✔ nested, scores | +| User posts/comments | `user NAME` | ✔ | ✔ | +| Backend + rate limit | `doctor` | ✔ | ✔ | + +## Procedure + +① `doctor` once per task if you have not called it this session — it tells you which +backend is live and how many seconds remain in the anonymous window. + +② Plan your calls before making them. Anonymous Reddit allows roughly **one request per +minute per IP**; the script sleeps until the window resets on a 429 and retries once, so +a five-call plan costs about five minutes. Prefer one `search --sub` over several `sub` +listings, and read one thread rather than the whole listing. + +③ For "what is the community saying" questions, read the thread bodies (`thread`) rather +than stopping at titles; the listing only carries the first ~300 characters of each post. + +④ Cite the permalink (`url` field), not the listing page, when the result feeds a report. +`grounded-citations` registers these URLs like any other source. + +⑤ If the user needs sustained Reddit access (monitoring, more than ~10 calls), stop and +ask them to add the OAuth credentials rather than grinding through the throttle. + +## Pitfalls + +- `www.reddit.com/…/.json`, `api.reddit.com` and `old.reddit.com` return 403 or an + empty "Welcome to Reddit" shell for datacentre IPs. Do not fall back to them; do not + spoof a browser User-Agent (also 403). +- `r.jina.ai` and the `browser_navigate` tool hit the same block ("blocked by network + security" / humanity check). `blocked-page-recovery`'s Wayback route can still recover + an **old** thread that was archived; it cannot fetch fresh ones. +- Anonymous thread feeds only contain the post plus top-level comments (Reddit caps the + feed at a handful of entries); scores and reply nesting are OAuth-only. +- Reddit's `limit` on feeds is advisory — expect 5–25 entries regardless of what you ask. +- Never paste `REDDIT_CLIENT_SECRET` into a chat or log; the script reads it from the + environment only. + +## Verification + +`python3 scripts/reddit.py doctor` prints `anonymous_feed: ok` and an +`x-ratelimit-reset` value; `sub announcements --limit 1` returns one entry with a +`reddit.com/r/announcements/comments/` URL. With credentials set, `doctor` prints +`active_backend: oauth` and `thread …` output shows numeric scores. diff --git a/website/docs/user-guide/skills/optional/research/research-blogwatcher.md b/website/docs/user-guide/skills/optional/research/research-blogwatcher.md index 84d9ac28aa..e929ea0863 100644 --- a/website/docs/user-guide/skills/optional/research/research-blogwatcher.md +++ b/website/docs/user-guide/skills/optional/research/research-blogwatcher.md @@ -15,7 +15,7 @@ Monitor blogs and RSS/Atom feeds via blogwatcher-cli tool. | | | |---|---| | Source | Optional — install with `hermes skills install official/research/blogwatcher` | -| Path | `optional-skills/research\blogwatcher` | +| Path | `optional-skills/research/blogwatcher` | | Version | `2.0.0` | | Author | JulienTant (fork of Hyaxia/blogwatcher) | | License | MIT | @@ -39,6 +39,7 @@ Track blog and RSS/Atom feed updates with the `blogwatcher-cli` tool. Supports a - **Recurring watch — use the cronjob tool's `monitor` field, not a bare schedule.** `monitor` runs a script each tick and only wakes the agent when output changes: set it to a script that runs `blogwatcher-cli scan >/dev/null 2>&1 && blogwatcher-cli articles` (deterministic output; new articles = changed output = agent wakes with the diff injected). Unchanged ticks cost zero LLM calls. Set `deliver` to route digests to a chat/channel; add `continuity: true` so consecutive digests can dedupe. - **Reading an article the user asks about**: `web_extract([url])` on the article URL from `blogwatcher-cli articles` — do not re-scrape by hand. - **One-off "watch this page for changes" without feed semantics**: skip this skill; the cronjob tool's `monitor` field accepts an http(s) URL directly. +- **One-off read of a feed or a site's latest posts, nothing to install**: the bundled `rss-feeds` skill (`scripts/feed.py read URL`); blogwatcher earns its install when you track many feeds with read/unread state. - **Company/competitor tracking with analysis and citations**: prefer the `competitor-news-monitor` skill; blogwatcher is the lighter raw-feed layer it can sit on. ## Installation diff --git a/website/sidebars.ts b/website/sidebars.ts index a50277bae6..fbe83d78c6 100644 --- a/website/sidebars.ts +++ b/website/sidebars.ts @@ -264,6 +264,7 @@ const sidebars: SidebarsConfig = { 'user-guide/skills/bundled/research/research-competitor-news-monitor', 'user-guide/skills/bundled/research/research-grounded-citations', 'user-guide/skills/bundled/research/research-llm-wiki', + 'user-guide/skills/bundled/research/research-rss-feeds', ], }, { @@ -272,6 +273,7 @@ const sidebars: SidebarsConfig = { key: 'skills-bundled-social-media', collapsed: true, items: [ + 'user-guide/skills/bundled/social-media/social-media-reddit-reading', 'user-guide/skills/bundled/social-media/social-media-xurl', ], }, From ee5b5ec21e576ccf9b941f9ff71330418415a5cb Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 12:49:04 -0700 Subject: [PATCH 053/276] docs(skills): reddit-reading spells out that no login is needed and OAuth is app-only The default backend needs no Reddit account, login, cookie or API key; the optional upgrade is a free 'script' app registration using the app-only client_credentials grant, never a user login. Said in the skill's Prerequisites (with a comparison table), Procedure, Pitfalls, the runtime doctor/thread notes, and a delimited .env.example block. --- .env.example | 9 +++++ skills/social-media/reddit-reading/SKILL.md | 35 +++++++++++++++---- .../reddit-reading/scripts/reddit.py | 14 +++++--- .../social-media-reddit-reading.md | 35 +++++++++++++++---- 4 files changed, 74 insertions(+), 19 deletions(-) diff --git a/.env.example b/.env.example index 02985b54a4..78c0108e1c 100644 --- a/.env.example +++ b/.env.example @@ -533,3 +533,12 @@ IMAGE_TOOLS_DEBUG=false # GOOGLE_CHAT_ALLOW_ALL_USERS=false # Set true to skip the allowlist # GOOGLE_CHAT_HOME_CHANNEL= # Default space (spaces/XXXX) for cron delivery # GOOGLE_CHAT_HOME_CHANNEL_NAME= # Display name for the home channel + +# ============================================================================= +# reddit-reading skill (optional) — app-only credentials, NOT a user login +# ============================================================================= +# The skill works with no credentials via Reddit's public feeds (~1 request/minute). +# For faster access with scores and nested comments, register a free "script" app +# at https://www.reddit.com/prefs/apps and paste its id and secret here. +# REDDIT_CLIENT_ID= +# REDDIT_CLIENT_SECRET= diff --git a/skills/social-media/reddit-reading/SKILL.md b/skills/social-media/reddit-reading/SKILL.md index 8d97c6ece2..7cc4e1f163 100644 --- a/skills/social-media/reddit-reading/SKILL.md +++ b/skills/social-media/reddit-reading/SKILL.md @@ -29,18 +29,32 @@ backend routing in [Agent Reach](https://github.com/Panniantong/Agent-Reach). ## Prerequisites -None for the anonymous path. For anything beyond a handful of calls per task, create -a free Reddit "script" app at https://www.reddit.com/prefs/apps and put the two values -in `~/.hermes/.env`: +**None.** No Reddit account, login, cookie, or API key is needed. The default backend is +Reddit's public Atom feeds (`.rss` endpoints), the only unauthenticated route Reddit still +serves to non-residential IPs. It is throttled to about one request per minute per IP and +returns thinner data (no scores, top-level comments only), which is fine for a few calls. + +**Optional upgrade (app credentials, still no user login):** for sustained use or full +data, register a free "script" type app at https://www.reddit.com/prefs/apps and put its +two values in `~/.hermes/.env`: ``` REDDIT_CLIENT_ID=... REDDIT_CLIENT_SECRET=... ``` -The script picks the OAuth backend automatically when both are set (~100 requests per -minute, scores, comment nesting, `num_comments`). Without them it uses Reddit's Atom -feeds, which are the only unauthenticated endpoints still served to non-residential IPs. +This is an application registration, not a login: the script uses the app-only +`client_credentials` grant, never a username, password, or browser cookie, and never +acts as the user. With both values set it switches to the OAuth API automatically +(~100 requests per minute, scores, nested comments, `num_comments`); if they are +missing or rejected it falls back to the anonymous feeds and says so on stderr. + +| | Anonymous feeds (default) | OAuth app credentials | +|---|---|---| +| Setup | nothing | 1-minute app registration, two `.env` values | +| Rate limit | ~1 request / minute / IP | ~100 requests / minute | +| Thread data | post + top-level comments, no scores | nested comments, scores, comment counts | +| Acts as a user | no | no | ## How to Run @@ -57,6 +71,8 @@ python3 scripts/reddit.py --json search "topic" # machine-reada ## Quick Reference +Every command works on both backends; the script chooses the backend, you never pass a flag. + | Need | Command | Anonymous | OAuth | |---|---|---|---| | Subreddit front page | `sub NAME --sort hot\|new\|top\|rising [--time week]` | ✔ | ✔ | @@ -83,7 +99,9 @@ than stopping at titles; the listing only carries the first ~300 characters of e `grounded-citations` registers these URLs like any other source. ⑤ If the user needs sustained Reddit access (monitoring, more than ~10 calls), stop and -ask them to add the OAuth credentials rather than grinding through the throttle. +ask them to register the app credentials (Prerequisites) rather than grinding through the +throttle. Tell them plainly: it is a free app registration, not logging Hermes into their +account. Never ask for a Reddit password or browser cookies. ## Pitfalls @@ -98,6 +116,9 @@ ask them to add the OAuth credentials rather than grinding through the throttle. - Reddit's `limit` on feeds is advisory — expect 5–25 entries regardless of what you ask. - Never paste `REDDIT_CLIENT_SECRET` into a chat or log; the script reads it from the environment only. +- Do not "fix" a 429 by retrying in a loop or adding a proxy; the throttle is per IP and + the script already waits out the window once. More than one 429 in a row means the + task needs the app credentials. ## Verification diff --git a/skills/social-media/reddit-reading/scripts/reddit.py b/skills/social-media/reddit-reading/scripts/reddit.py index 3f00646437..85769c9f85 100644 --- a/skills/social-media/reddit-reading/scripts/reddit.py +++ b/skills/social-media/reddit-reading/scripts/reddit.py @@ -4,8 +4,10 @@ Two backends, chosen automatically: * **OAuth API** (preferred when ``REDDIT_CLIENT_ID`` + ``REDDIT_CLIENT_SECRET`` are set): - a free "script" app from https://www.reddit.com/prefs/apps; ~100 requests/minute, - full JSON including scores and nested comments. + app-only ``client_credentials`` grant for a free "script" app registered at + https://www.reddit.com/prefs/apps. No username, password or cookie is ever used and + the script never acts as a user. ~100 requests/minute, full JSON including scores + and nested comments. * **Anonymous Atom feeds** (``.rss`` endpoints): the only unauthenticated path Reddit still serves to server IPs (``.json`` and old.reddit return 403 / an empty shell). Roughly ONE request per minute per IP; the script sleeps until the window resets @@ -181,7 +183,8 @@ def atom_thread(sub: str, post_id: str, limit: int) -> dict: raise SystemExit("thread feed returned no entries") post, comments = entries[0], entries[1:] post["comments"] = [{"author": c["author"], "created": c["created"], "body": c["body"], "url": c["url"]} for c in comments] - post["note"] = "anonymous feed: scores and nesting unavailable; set REDDIT_CLIENT_ID/SECRET for full data" + post["note"] = ("anonymous feed: scores and nesting unavailable; register a free Reddit script app and set " + "REDDIT_CLIENT_ID/REDDIT_CLIENT_SECRET (no user login) for full data") return post @@ -241,9 +244,10 @@ def cmd_doctor(a, token): except urllib.error.HTTPError as exc: report["anonymous_feed"] = f"HTTP {exc.code}" report["notes"] = [ - "anonymous .rss: ~1 request/minute per IP (x-ratelimit-remaining drops to 0 after each call)", + "anonymous .rss needs no account, login, cookie or key; ~1 request/minute per IP", "www.reddit.com .json, api.reddit.com and old.reddit are 403 / an empty shell for server IPs", - "for more than a few calls per task create a free script app and export REDDIT_CLIENT_ID/SECRET", + "for more than a few calls per task register a free 'script' app at reddit.com/prefs/apps and set " + "REDDIT_CLIENT_ID/REDDIT_CLIENT_SECRET in .env (app-only credentials; Hermes never logs in as the user)", ] return report diff --git a/website/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading.md b/website/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading.md index 219195c947..812e9d89b9 100644 --- a/website/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading.md +++ b/website/docs/user-guide/skills/bundled/social-media/social-media-reddit-reading.md @@ -47,18 +47,32 @@ backend routing in [Agent Reach](https://github.com/Panniantong/Agent-Reach). ## Prerequisites -None for the anonymous path. For anything beyond a handful of calls per task, create -a free Reddit "script" app at https://www.reddit.com/prefs/apps and put the two values -in `~/.hermes/.env`: +**None.** No Reddit account, login, cookie, or API key is needed. The default backend is +Reddit's public Atom feeds (`.rss` endpoints), the only unauthenticated route Reddit still +serves to non-residential IPs. It is throttled to about one request per minute per IP and +returns thinner data (no scores, top-level comments only), which is fine for a few calls. + +**Optional upgrade (app credentials, still no user login):** for sustained use or full +data, register a free "script" type app at https://www.reddit.com/prefs/apps and put its +two values in `~/.hermes/.env`: ``` REDDIT_CLIENT_ID=... REDDIT_CLIENT_SECRET=... ``` -The script picks the OAuth backend automatically when both are set (~100 requests per -minute, scores, comment nesting, `num_comments`). Without them it uses Reddit's Atom -feeds, which are the only unauthenticated endpoints still served to non-residential IPs. +This is an application registration, not a login: the script uses the app-only +`client_credentials` grant, never a username, password, or browser cookie, and never +acts as the user. With both values set it switches to the OAuth API automatically +(~100 requests per minute, scores, nested comments, `num_comments`); if they are +missing or rejected it falls back to the anonymous feeds and says so on stderr. + +| | Anonymous feeds (default) | OAuth app credentials | +|---|---|---| +| Setup | nothing | 1-minute app registration, two `.env` values | +| Rate limit | ~1 request / minute / IP | ~100 requests / minute | +| Thread data | post + top-level comments, no scores | nested comments, scores, comment counts | +| Acts as a user | no | no | ## How to Run @@ -75,6 +89,8 @@ python3 scripts/reddit.py --json search "topic" # machine-reada ## Quick Reference +Every command works on both backends; the script chooses the backend, you never pass a flag. + | Need | Command | Anonymous | OAuth | |---|---|---|---| | Subreddit front page | `sub NAME --sort hot\|new\|top\|rising [--time week]` | ✔ | ✔ | @@ -101,7 +117,9 @@ than stopping at titles; the listing only carries the first ~300 characters of e `grounded-citations` registers these URLs like any other source. ⑤ If the user needs sustained Reddit access (monitoring, more than ~10 calls), stop and -ask them to add the OAuth credentials rather than grinding through the throttle. +ask them to register the app credentials (Prerequisites) rather than grinding through the +throttle. Tell them plainly: it is a free app registration, not logging Hermes into their +account. Never ask for a Reddit password or browser cookies. ## Pitfalls @@ -116,6 +134,9 @@ ask them to add the OAuth credentials rather than grinding through the throttle. - Reddit's `limit` on feeds is advisory — expect 5–25 entries regardless of what you ask. - Never paste `REDDIT_CLIENT_SECRET` into a chat or log; the script reads it from the environment only. +- Do not "fix" a 429 by retrying in a loop or adding a proxy; the throttle is per IP and + the script already waits out the window once. More than one 429 in a row means the + task needs the app credentials. ## Verification From 96ed0e71ea2d75d325cc05c3e6468d213de2408d Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Sat, 5 Sep 2026 13:26:03 -0600 Subject: [PATCH 054/276] fix(bot-mode): match scoped bot selection in reset guard --- .../hermes-bots/plugin.mentions.test.ts | 54 ++++++++++++++++++- .../src/plugins/hermes-bots/plugin.tsx | 3 +- 2 files changed, 54 insertions(+), 3 deletions(-) diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.mentions.test.ts b/apps/desktop/src/plugins/hermes-bots/plugin.mentions.test.ts index de02c50025..bc05ef6433 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.mentions.test.ts +++ b/apps/desktop/src/plugins/hermes-bots/plugin.mentions.test.ts @@ -21,6 +21,8 @@ import { beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' +import type { RosterRow } from './types' + interface MentionCompletionItem { display: string insert: string @@ -58,7 +60,7 @@ const ROSTER = { } const { cache, hostMock, live } = vi.hoisted(() => { - const live = { focused: 'default', profile: 'default' } + const live = { focused: 'default', profile: 'default', stored: null as string | null } return { cache: new Map(), @@ -69,7 +71,7 @@ const { cache, hostMock, live } = vi.hoisted(() => { state: { connectionId: { get: () => 'local', listen: () => () => undefined }, focusedSessionProfile: { get: () => live.focused, listen: () => () => undefined }, - focusedStoredSessionId: { get: () => null, listen: () => () => undefined }, + focusedStoredSessionId: { get: () => live.stored, listen: () => () => undefined }, gateway: { get: () => null, listen: () => () => undefined }, profile: { get: () => live.profile, listen: () => () => undefined } } @@ -173,6 +175,54 @@ beforeEach(() => { vi.clearAllMocks() live.focused = 'default' live.profile = 'default' + live.stored = null +}) + +describe('Bot Chat reset guard', () => { + it.each([ + { connectionId: 'local', sourceScoped: true }, + { connectionId: 'remote', remoteSource: true, sourceScoped: true }, + {} + ])('compacts only the selected bot canonical chat: %j', async source => { + const { handler } = await contributions() + const { $lastRoster } = await import('./data') + const { saveSelectedRosterBot } = await import('./bot-state') + + const selected: RosterRow = { + ...source, + name: 'writer', + canonical_session: { id: 'bot-root', resolved_id: 'bot-tip' } + } + + // An identically named bot on another connection must not win the lookup. + $lastRoster.set([ + { name: 'writer', connectionId: 'other', sourceScoped: true, canonical_session: { id: 'other-chat' } }, + selected + ]) + saveSelectedRosterBot(selected) + + for (const stored of ['bot-root', 'bot-tip']) { + live.stored = stored + + for (const text of ['/new', '/reset']) { + hostMock.notify.mockClear() + expect(await handler({ text })).toEqual({ text: '/compact' }) + expect(hostMock.notify).toHaveBeenCalledOnce() + expect(hostMock.notify).toHaveBeenCalledWith(expect.objectContaining({ title: 'This chat never resets' })) + } + } + + for (const stored of ['scratch-chat', 'other-chat', null]) { + live.stored = stored + + for (const text of ['/new', '/reset']) { + hostMock.notify.mockClear() + const draft = { text } + expect(await handler(draft)).toBe(draft) + expect(hostMock.notify).not.toHaveBeenCalled() + } + } + }) }) describe('@-mention completions', () => { diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.tsx b/apps/desktop/src/plugins/hermes-bots/plugin.tsx index ffdcaeb8fa..761776fa23 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.tsx +++ b/apps/desktop/src/plugins/hermes-bots/plugin.tsx @@ -37,6 +37,7 @@ import { $lastRoster, botHandle, botMentionTag, + botSelectionKey, cachedUnionRoster, isActiveRosterBot, migrateBotMeta, @@ -665,7 +666,7 @@ export default { // server-side by name), matching either the durable row id or // the compression-lineage tip currently on screen. const roster = $lastRoster.get() - const row = Array.isArray(roster) ? roster.find(bot => bot?.name === activeBot) : null + const row = Array.isArray(roster) ? roster.find(bot => botSelectionKey(bot) === activeBot) : null // The STORED id, which is the id space canonical_session is keyed // in. `host.state.activeSessionId` is the runtime id and could From 3bc1780f2d3ec3df795a3d205e84f63c0fa768a8 Mon Sep 17 00:00:00 2001 From: Muno <90378186+Muno459@users.noreply.github.com> Date: Mon, 4 May 2026 14:25:49 +0200 Subject: [PATCH 055/276] fix(toolsets): merge MCP tools when alias collides with static toolset MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit If an MCP server registers itself under the same name as a built-in toolset (e.g. an MCP "homeassistant" running alongside the built-in `homeassistant` toolset), `get_toolset()` returned the static definition early and the MCP tools were silently shadowed — they got registered into `mcp-homeassistant` but never surfaced when the agent looked up `homeassistant`. Detect the collision via the registry alias (`alias_target` starting with `mcp-`) and merge the two tool lists. The description is annotated so the source is visible in `/toolsets info`. --- tests/test_toolsets.py | 18 ++++++++++++++++++ toolsets.py | 10 ++++++++-- 2 files changed, 26 insertions(+), 2 deletions(-) diff --git a/tests/test_toolsets.py b/tests/test_toolsets.py index 944d96b742..daf06b469b 100644 --- a/tests/test_toolsets.py +++ b/tests/test_toolsets.py @@ -56,6 +56,24 @@ class TestGetToolset: assert set(ts["tools"]) == {"web_search", "web_extract", "web_search_plus"} + def test_static_and_mcp_alias_with_same_name_are_merged(self, monkeypatch): + # An MCP server named like a built-in toolset registers a bare alias to its + # `mcp-` toolset; the static entry must union those tools in (and keep + # its own includes) instead of shadowing the server. + TOOLSETS["_mergetest"] = {"description": "static", "tools": ["builtin_tool_a"], "includes": ["web"]} + try: + reg = ToolRegistry() + reg.register(name="mcp__mergetest_call", toolset="mcp-_mergetest", + schema=_make_schema("mcp__mergetest_call", "Call"), handler=_dummy_handler) + reg.register_toolset_alias("_mergetest", "mcp-_mergetest") + monkeypatch.setattr("tools.registry.registry", reg) + + ts = get_toolset("_mergetest") + assert {"builtin_tool_a", "mcp__mergetest_call"} <= set(ts["tools"]) + assert ts["includes"] == ["web"] + finally: + del TOOLSETS["_mergetest"] + class TestResolveToolset: def test_leaf_toolset(self): diff --git a/toolsets.py b/toolsets.py index cf7aef5c5a..ffade803fa 100644 --- a/toolsets.py +++ b/toolsets.py @@ -281,8 +281,14 @@ def get_toolset(name: str, *, include_registry: bool = True) -> Optional[Dict[st return toolset if toolset else None if toolset: - merged_tools = sorted(set(toolset.get("tools", [])) | set(registry.get_tool_names_for_toolset(name))) - return {**toolset, "tools": merged_tools} + merged_tools = set(toolset.get("tools", [])) | set(registry.get_tool_names_for_toolset(name)) + # An MCP server named like a built-in toolset ("homeassistant", "browser") registers a bare + # alias to its `mcp-` toolset; without this union the static entry shadows it and the + # server's tools never reach the model even though discovery registered them. + alias_target = registry.get_toolset_alias_target(name) + if alias_target and alias_target != name: + merged_tools |= set(registry.get_tool_names_for_toolset(alias_target)) + return {**toolset, "tools": sorted(merged_tools)} if name in _get_plugin_toolset_names(): # Plugin toolset; shown as its MCP server alias when one exists. From 3f880550352b09f000f4bcaca0d4472a883ba5bc Mon Sep 17 00:00:00 2001 From: Muno459 <90378186+Muno459@users.noreply.github.com> Date: Mon, 13 Jul 2026 00:43:20 +0200 Subject: [PATCH 056/276] test(mcp): integration regression for colliding static toolset name via _register_server_tools --- tests/tools/test_mcp_dynamic_discovery.py | 28 +++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/tests/tools/test_mcp_dynamic_discovery.py b/tests/tools/test_mcp_dynamic_discovery.py index 948560d8ea..76fcb907a4 100644 --- a/tests/tools/test_mcp_dynamic_discovery.py +++ b/tests/tools/test_mcp_dynamic_discovery.py @@ -36,6 +36,34 @@ class TestRegisterServerTools: assert validate_toolset("my_srv") is True assert "mcp__my_srv__my_tool" in resolve_toolset("my_srv") + def test_colliding_static_toolset_name_merges_both_tool_sets(self, mock_registry): + """An MCP server named after a built-in toolset must not be shadowed. + + Regression: an MCP server registered as `homeassistant` (colliding + with the static `homeassistant` toolset) had its tools silently + dropped because get_toolset() returned the static definition without + consulting the alias registered by _register_server_tools(). + """ + from toolsets import TOOLSETS, get_toolset, resolve_toolset + + assert "homeassistant" in TOOLSETS # collision premise + static_tools = set(TOOLSETS["homeassistant"]["tools"]) + + server = MCPServerTask("homeassistant") + server._tools = [_make_mcp_tool("get_entities", "List HA entities")] + server.session = MagicMock() + + with patch("tools.registry.registry", mock_registry): + registered = _register_server_tools("homeassistant", server, {}) + assert "mcp__homeassistant__get_entities" in registered + + ts = get_toolset("homeassistant") + # Static built-ins are still present... + assert static_tools <= set(ts["tools"]) + # ...and the MCP server's tools are no longer shadowed. + assert "mcp__homeassistant__get_entities" in ts["tools"] + assert "mcp__homeassistant__get_entities" in resolve_toolset("homeassistant") + class TestRefreshTools: """Tests for MCPServerTask._refresh_tools nuke-and-repave cycle.""" From 245e48008fa814b3251f50755eb656bd9fb86cb1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:31:16 -0700 Subject: [PATCH 057/276] fix: listing surfaces show the merged view for MCP servers named like built-in toolsets get_all_toolsets() copied static TOOLSETS entries verbatim, so /toolsets and the dashboard showed only the built-in tools for a colliding name even though get_toolset() now unions the MCP alias in. Route those names through get_toolset() and document the alias/collision behaviour. --- toolsets.py | 4 ++++ website/docs/reference/toolsets-reference.md | 2 +- 2 files changed, 5 insertions(+), 1 deletion(-) diff --git a/toolsets.py b/toolsets.py index ffade803fa..d4a33dcf53 100644 --- a/toolsets.py +++ b/toolsets.py @@ -414,10 +414,14 @@ def _plugin_display_names() -> List[str]: def get_all_toolsets() -> Dict[str, Dict[str, Any]]: """All toolset definitions: static plus plugin-registered.""" result = dict(TOOLSETS) + aliases = _get_registry_toolset_aliases() for display_name in _plugin_display_names(): toolset = None if display_name in result else get_toolset(display_name) if toolset: result[display_name] = toolset + # Static names an MCP server also aliases show the merged view get_toolset() resolves. + for name in TOOLSETS.keys() & aliases.keys(): + result[name] = get_toolset(name) or result[name] return result diff --git a/website/docs/reference/toolsets-reference.md b/website/docs/reference/toolsets-reference.md index 5904f1a9f7..c6925facb0 100644 --- a/website/docs/reference/toolsets-reference.md +++ b/website/docs/reference/toolsets-reference.md @@ -132,7 +132,7 @@ mcp_servers: args: ["-y", "@modelcontextprotocol/server-github"] ``` -This creates a `mcp-github` toolset you can reference in `--toolsets` or platform configs. +This creates a `mcp-github` toolset you can reference in `--toolsets` or platform configs. The bare server name (`github`) works as an alias. If a server is named like a built-in toolset (`homeassistant`, `browser`), that name resolves to the built-in tools **plus** the server's `mcp____*` tools; neither side shadows the other. ### Plugin toolsets From a483d7742fdfe37657a8372d065a635db5b25e71 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 13:06:31 +0530 Subject: [PATCH 058/276] chore: map ruichenzhou@outlook.com to @Zhou-Ruichen (#77933 salvage) --- contributors/emails/ruichenzhou@outlook.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/ruichenzhou@outlook.com diff --git a/contributors/emails/ruichenzhou@outlook.com b/contributors/emails/ruichenzhou@outlook.com new file mode 100644 index 0000000000..1cf5dccea0 --- /dev/null +++ b/contributors/emails/ruichenzhou@outlook.com @@ -0,0 +1,2 @@ +Zhou-Ruichen +# PR #77933 salvage (ssh: probe-only prompt backend probe) From 50b54bb18f0f9862da0fb7de630183bff0fa88a3 Mon Sep 17 00:00:00 2001 From: Ruichen Zhou Date: Tue, 4 Aug 2026 02:45:19 +0800 Subject: [PATCH 059/276] fix(ssh): isolate prompt backend probes from file sync The prompt-time backend probe built a normal SSHEnvironment just to run a one-line `uname`. That constructor detects the remote home, creates the ~/.hermes tree, force-uploads every sync file and snapshots a login session; when the throwaway object was later garbage-collected, __del__ -> cleanup() ran sync_back() and `ssh -O exit` against the ControlMaster socket the agent's real environment shares (keyed by user@host:port). Add an internal probe_only construction path for SSH: an isolated, same-length ControlMaster socket (keyed by the instance's session id), no remote dir setup, no FileSyncManager, no session snapshot. The probe now tears its own connection down explicitly on success, non-zero exit and exception, without replacing the probe result when cleanup fails. Normal SSH callers and non-SSH backends are unchanged. Salvaged from #77933 onto the facade/sibling layout (the probe body moved to _run_backend_probe, _create_environment to tools/terminal_tool_backends.py). --- agent/prompt_builder.py | 15 ++++---- tests/agent/test_prompt_builder.py | 44 ++++++++++++++++++---- tests/tools/test_ssh_environment.py | 58 +++++++++++++++++++++++++++++ tools/environments/ssh.py | 21 +++++++++-- tools/terminal_tool_backends.py | 11 +++--- 5 files changed, 125 insertions(+), 24 deletions(-) diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index 69c7a72906..eb65ad3563 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -876,19 +876,20 @@ def _run_backend_probe(env_type: str, terminal_tool) -> str: container_config=(_container_config_from_config(config) if terminal_tool._is_container_backend(env_type) else None), task_id="prompt-backend-probe", host_cwd=config.get("host_cwd"), + # ssh: an isolated ControlMaster socket and no remote dir setup / file sync / snapshot — + # a normal SSHEnvironment would upload the whole ~/.hermes tree just to run `uname`, and + # its later __del__ would sync_back() and close the master shared with the agent's own env. + probe_only=env_type == "ssh", ) try: result = env.execute(_BACKEND_PROBE_CMD, timeout=4) finally: # One-shot `uname`; without teardown the backend leaves a second idle sandbox # (task_id="prompt-backend-probe") running for the whole process next to the agent's own. - # ssh is left alone: no task-scoped sandbox, and its cleanup() closes a ControlMaster socket - # (keyed by user@host:port) shared with the agent's real environment; ControlPersist expires it. - if env_type != "ssh": - try: - _cleanup_env(env, force_remove=True) - except Exception: - logger.debug("Backend probe cleanup failed", exc_info=True) + try: + _cleanup_env(env, force_remove=True) + except Exception: + logger.debug("Backend probe cleanup failed", exc_info=True) if result.get("returncode") != 0: logger.debug("Backend probe returned non-zero: %r", result) return "" diff --git a/tests/agent/test_prompt_builder.py b/tests/agent/test_prompt_builder.py index 0e23a75257..a3dd7becc7 100644 --- a/tests/agent/test_prompt_builder.py +++ b/tests/agent/test_prompt_builder.py @@ -932,18 +932,20 @@ class TestEnvironmentHints: assert _pb._probe_remote_backend("singularity") is not None assert calls == ["bare"] - def test_probe_remote_backend_does_not_tear_down_ssh(self, monkeypatch): - """SSH has no task-scoped sandbox: its cleanup() closes a ControlMaster - socket shared with the agent's real environment, so the probe must - leave it alone (nothing leaks — ControlPersist expires the master).""" + def test_probe_remote_backend_ssh_is_probe_only_and_torn_down(self, monkeypatch): + """SSH probe: a normal SSHEnvironment would create remote dirs, force-upload + ~/.hermes and snapshot a session just to run `uname`, and its __del__ would + later sync_back() and close the ControlMaster shared with the agent's real + environment. The probe must request a probe-only instance (own socket, no + setup/sync) and tear it down itself.""" import agent.prompt_builder as _pb monkeypatch.setenv("TERMINAL_ENV", "ssh") _pb._clear_backend_probe_cache() - calls = [] + created, calls = {}, [] - class _SharedSshEnv: + class _ProbeSshEnv: def execute(self, cmd, timeout=None): return { "returncode": 0, @@ -957,10 +959,36 @@ class TestEnvironmentHints: calls.append("cleanup") import tools.terminal_tool_backends as _tt - monkeypatch.setattr(_tt, "_create_environment", lambda **kw: _SharedSshEnv()) + + def _fake_create(**kw): + created.update(kw) + return _ProbeSshEnv() + + monkeypatch.setattr(_tt, "_create_environment", _fake_create) assert _pb._probe_remote_backend("ssh") is not None - assert calls == [] + assert created["probe_only"] is True + assert calls == ["cleanup"] + + def test_probe_remote_backend_ssh_cleanup_error_keeps_result(self, monkeypatch): + """A failing teardown of the throwaway probe connection must not discard the + metadata the probe already collected.""" + import agent.prompt_builder as _pb + + monkeypatch.setenv("TERMINAL_ENV", "ssh") + _pb._clear_backend_probe_cache() + + class _Env: + def execute(self, cmd, timeout=None): + return {"returncode": 0, "output": "os=Linux\nkernel=6.8.0\nhome=/h\ncwd=/h\nuser=u\n"} + + def cleanup(self): + raise RuntimeError("cleanup failed") + + import tools.terminal_tool_backends as _tt + monkeypatch.setattr(_tt, "_create_environment", lambda **kw: _Env()) + + assert "Linux 6.8.0" in _pb._probe_remote_backend("ssh") def test_environment_hint_from_env_var_is_appended(self, monkeypatch): diff --git a/tests/tools/test_ssh_environment.py b/tests/tools/test_ssh_environment.py index 9c99f25cb6..a05975429c 100644 --- a/tests/tools/test_ssh_environment.py +++ b/tests/tools/test_ssh_environment.py @@ -233,6 +233,64 @@ class TestSSHPreflight: assert env.user == "alice" +@pytest.fixture +def _mock_ssh_runtime(monkeypatch, tmp_path): + hooks = { + "_establish_connection": MagicMock(), + "_detect_remote_home": MagicMock(return_value="/home/alice"), + "_ensure_remote_dirs": MagicMock(), + "init_session": MagicMock(), + } + monkeypatch.setattr(ssh_env.tempfile, "gettempdir", lambda: str(tmp_path)) + monkeypatch.setattr(ssh_env.shutil, "which", lambda _name: "/usr/bin/ssh") + for name, hook in hooks.items(): + monkeypatch.setattr(ssh_env.SSHEnvironment, name, hook) + hooks["sync_factory"] = MagicMock(return_value=MagicMock()) + monkeypatch.setattr(ssh_env, "FileSyncManager", hooks["sync_factory"]) + return hooks + + +class TestSSHProbeOnly: + def test_probe_only_skips_state_sync_and_session_setup(self, _mock_ssh_runtime): + env = ssh_env.SSHEnvironment(host="example.com", user="alice", probe_only=True) + env._before_execute() + env.cleanup() + + _mock_ssh_runtime["_establish_connection"].assert_called_once_with() + _mock_ssh_runtime["_detect_remote_home"].assert_not_called() + _mock_ssh_runtime["_ensure_remote_dirs"].assert_not_called() + _mock_ssh_runtime["sync_factory"].assert_not_called() + _mock_ssh_runtime["init_session"].assert_not_called() + assert env._sync_manager is None + + def test_probe_only_control_socket_is_isolated(self, monkeypatch, _mock_ssh_runtime): + control_exit_calls = [] + + def _fake_run(*args, **kwargs): + control_exit_calls.append(args[0]) + return subprocess.CompletedProcess([], 0) + + monkeypatch.setattr(ssh_env.subprocess, "run", _fake_run) + + normal = ssh_env.SSHEnvironment(host="example.com", user="alice") + first_probe = ssh_env.SSHEnvironment(host="example.com", user="alice", probe_only=True) + second_probe = ssh_env.SSHEnvironment(host="example.com", user="alice", probe_only=True) + + assert first_probe.control_socket != normal.control_socket + assert second_probe.control_socket != first_probe.control_socket + assert len(first_probe.control_socket.name) == len(normal.control_socket.name) + + normal.control_socket.touch() + first_probe.control_socket.touch() + first_probe.cleanup() + first_probe.cleanup() + + assert normal.control_socket.exists() + assert not first_probe.control_socket.exists() + assert len(control_exit_calls) == 1 + normal.cleanup() + + def _setup_ssh_env(monkeypatch, persistent: bool): monkeypatch.setenv("TERMINAL_ENV", "ssh") monkeypatch.setenv("TERMINAL_SSH_HOST", _SSH_HOST) diff --git a/tools/environments/ssh.py b/tools/environments/ssh.py index 04f2a6e43b..3c97735df2 100644 --- a/tools/environments/ssh.py +++ b/tools/environments/ssh.py @@ -48,6 +48,7 @@ class SSHEnvironment(BaseEnvironment): Spawn-per-call: every execute() spawns a fresh ``ssh ... bash -c`` process. Session snapshot preserves env vars across calls; CWD persists via in-band stdout markers. Uses SSH ControlMaster for connection reuse. + Probe-only instances use an isolated connection without sync or session state. """ # Passthrough values are re-forwarded on every command (see _run_bash), so like docker/local @@ -55,18 +56,29 @@ class SSHEnvironment(BaseEnvironment): _profile_scoped_passthrough = True def __init__(self, host: str, user: str, cwd: str = "~", - timeout: int = 60, port: int = 22, key_path: str = ""): + timeout: int = 60, port: int = 22, key_path: str = "", + probe_only: bool = False): super().__init__(cwd=cwd, timeout=timeout) self.host, self.user, self.port, self.key_path = host, user, port, key_path + self._probe_only = probe_only + self._sync_manager = None self.control_dir = Path(tempfile.gettempdir()) / "hermes-ssh" self.control_dir.mkdir(parents=True, exist_ok=True) # Short, deterministic socket name: the path must stay under macOS's 104-byte sun_path # limit (raw user@host:port + SSH's 16-byte suffix under a deep $TMPDIR exceeds it), and - # stability across reconnects keeps ControlMaster reuse working. - _socket_id = hashlib.sha256(f"{user}@{host}:{port}".encode()).hexdigest()[:16] + # stability across reconnects keeps ControlMaster reuse working. A probe gets its own + # per-instance socket so its cleanup() can never close the agent's shared master. + socket_key = f"{user}@{host}:{port}" + if self._probe_only: + socket_key = f"{socket_key}:probe:{self._session_id}" + _socket_id = hashlib.sha256(socket_key.encode()).hexdigest()[:16] self.control_socket = self.control_dir / f"{_socket_id}.sock" _ensure_ssh_available() self._establish_connection() + if self._probe_only: + self._remote_home = "" + return + self._remote_home = self._detect_remote_home() self._ensure_remote_dirs() self._sync_manager = FileSyncManager( @@ -247,7 +259,8 @@ class SSHEnvironment(BaseEnvironment): f"Remote file cleanup on {self.host}") def _before_execute(self) -> None: - self._sync_manager.sync() # rate-limited internally + if self._sync_manager is not None: + self._sync_manager.sync() # rate-limited internally def _run_bash(self, cmd_string: str, *, login: bool = False, timeout: int = 120, stdin_data: str | None = None) -> subprocess.Popen: diff --git a/tools/terminal_tool_backends.py b/tools/terminal_tool_backends.py index 63a3dd1925..335bb1c304 100644 --- a/tools/terminal_tool_backends.py +++ b/tools/terminal_tool_backends.py @@ -173,11 +173,11 @@ _build_daytona_env = functools.partial(_build_sandbox_env, "daytona") _build_vercel_env = functools.partial(_build_sandbox_env, "vercel_sandbox") -def _build_ssh_env(*, cwd, timeout, ssh_config, **_): +def _build_ssh_env(*, cwd, timeout, ssh_config, probe_only=False, **_): if not ssh_config or not ssh_config.get("host") or not ssh_config.get("user"): raise ValueError("SSH environment requires ssh_host and ssh_user to be configured") return _SSHEnvironment(host=ssh_config["host"], user=ssh_config["user"], port=ssh_config.get("port", 22), - key_path=ssh_config.get("key", ""), cwd=cwd, timeout=timeout) + key_path=ssh_config.get("key", ""), cwd=cwd, timeout=timeout, probe_only=probe_only) def _build_plugin_env(*, env_type, image, cwd, timeout, cc, task_id, **_): @@ -210,13 +210,14 @@ _ENV_BUILDERS = {"local": _build_local_env, "docker": _build_docker_env, "singul def _create_environment(env_type: str, image: str, cwd: str, timeout: int, ssh_config: dict = None, container_config: dict = None, local_config: dict = None, task_id: str = "default", - host_cwd: Optional[str] = None): + host_cwd: Optional[str] = None, probe_only: bool = False): """Create an execution environment (instance with ``execute()``) for *env_type*. ``image`` is ignored for local/ssh/vercel; ``container_config`` carries the container_*/docker_* resource keys; ``host_cwd`` is - the host dir bound into Docker when cwd mounting is enabled. Unknown types fall through to plugin backends.""" + the host dir bound into Docker when cwd mounting is enabled. ``probe_only`` asks ssh for a throwaway + connection with no remote setup/sync (the prompt-time probe). Unknown types fall through to plugin backends.""" builder = _ENV_BUILDERS.get(env_type, _build_plugin_env) return builder(env_type=env_type, image=image, cwd=cwd, timeout=timeout, cc=container_config or {}, - task_id=task_id, ssh_config=ssh_config, host_cwd=host_cwd) + task_id=task_id, ssh_config=ssh_config, host_cwd=host_cwd, probe_only=probe_only) # --- Requirement checkers: one generic path driven by _BACKEND_SPECS; optional fields, checked in order: From 9a84bee265daad14340a80d7585928cd8ea1f9eb Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 13:22:10 +0530 Subject: [PATCH 060/276] refactor(ssh): drop probe-only bookkeeping the probe never reads - probe_only is consumed inside __init__ only; no instance attribute. - _sync_manager is assigned only on the probe branch, after the connection is up, so a normal SSHEnvironment whose constructor fails still behaves exactly as before (its __del__ cleanup does not reach the shared socket teardown). - _remote_home has no reader on the probe path; not assigned. - prompt_builder passes probe_only=True unconditionally: which backends honor it is the builder table's decision, not a caller-side env_type check. - Tests trimmed to the probe_only contract: the cleanup-raises case was green on main (pre-existing try/except), the socket-name length and double-cleanup assertions covered pre-existing behaviour. --- agent/prompt_builder.py | 8 ++++---- tests/agent/test_prompt_builder.py | 21 --------------------- tests/tools/test_ssh_environment.py | 3 --- tools/environments/ssh.py | 10 +++------- 4 files changed, 7 insertions(+), 35 deletions(-) diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index eb65ad3563..afc79fcc22 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -876,10 +876,10 @@ def _run_backend_probe(env_type: str, terminal_tool) -> str: container_config=(_container_config_from_config(config) if terminal_tool._is_container_backend(env_type) else None), task_id="prompt-backend-probe", host_cwd=config.get("host_cwd"), - # ssh: an isolated ControlMaster socket and no remote dir setup / file sync / snapshot — - # a normal SSHEnvironment would upload the whole ~/.hermes tree just to run `uname`, and - # its later __del__ would sync_back() and close the master shared with the agent's own env. - probe_only=env_type == "ssh", + # Only ssh honors this: an isolated ControlMaster socket and no remote dir setup / file sync / + # snapshot. A normal SSHEnvironment would upload the whole ~/.hermes tree just to run `uname`, + # and its later __del__ would sync_back() and close the master shared with the agent's own env. + probe_only=True, ) try: result = env.execute(_BACKEND_PROBE_CMD, timeout=4) diff --git a/tests/agent/test_prompt_builder.py b/tests/agent/test_prompt_builder.py index a3dd7becc7..af57363092 100644 --- a/tests/agent/test_prompt_builder.py +++ b/tests/agent/test_prompt_builder.py @@ -970,27 +970,6 @@ class TestEnvironmentHints: assert created["probe_only"] is True assert calls == ["cleanup"] - def test_probe_remote_backend_ssh_cleanup_error_keeps_result(self, monkeypatch): - """A failing teardown of the throwaway probe connection must not discard the - metadata the probe already collected.""" - import agent.prompt_builder as _pb - - monkeypatch.setenv("TERMINAL_ENV", "ssh") - _pb._clear_backend_probe_cache() - - class _Env: - def execute(self, cmd, timeout=None): - return {"returncode": 0, "output": "os=Linux\nkernel=6.8.0\nhome=/h\ncwd=/h\nuser=u\n"} - - def cleanup(self): - raise RuntimeError("cleanup failed") - - import tools.terminal_tool_backends as _tt - monkeypatch.setattr(_tt, "_create_environment", lambda **kw: _Env()) - - assert "Linux 6.8.0" in _pb._probe_remote_backend("ssh") - - def test_environment_hint_from_env_var_is_appended(self, monkeypatch): """HERMES_ENVIRONMENT_HINT lets an embedder describe the runtime env.""" import agent.prompt_builder as _pb diff --git a/tests/tools/test_ssh_environment.py b/tests/tools/test_ssh_environment.py index a05975429c..44c687992f 100644 --- a/tests/tools/test_ssh_environment.py +++ b/tests/tools/test_ssh_environment.py @@ -261,7 +261,6 @@ class TestSSHProbeOnly: _mock_ssh_runtime["_ensure_remote_dirs"].assert_not_called() _mock_ssh_runtime["sync_factory"].assert_not_called() _mock_ssh_runtime["init_session"].assert_not_called() - assert env._sync_manager is None def test_probe_only_control_socket_is_isolated(self, monkeypatch, _mock_ssh_runtime): control_exit_calls = [] @@ -283,12 +282,10 @@ class TestSSHProbeOnly: normal.control_socket.touch() first_probe.control_socket.touch() first_probe.cleanup() - first_probe.cleanup() assert normal.control_socket.exists() assert not first_probe.control_socket.exists() assert len(control_exit_calls) == 1 - normal.cleanup() def _setup_ssh_env(monkeypatch, persistent: bool): diff --git a/tools/environments/ssh.py b/tools/environments/ssh.py index 3c97735df2..290b81af49 100644 --- a/tools/environments/ssh.py +++ b/tools/environments/ssh.py @@ -48,7 +48,6 @@ class SSHEnvironment(BaseEnvironment): Spawn-per-call: every execute() spawns a fresh ``ssh ... bash -c`` process. Session snapshot preserves env vars across calls; CWD persists via in-band stdout markers. Uses SSH ControlMaster for connection reuse. - Probe-only instances use an isolated connection without sync or session state. """ # Passthrough values are re-forwarded on every command (see _run_bash), so like docker/local @@ -60,8 +59,6 @@ class SSHEnvironment(BaseEnvironment): probe_only: bool = False): super().__init__(cwd=cwd, timeout=timeout) self.host, self.user, self.port, self.key_path = host, user, port, key_path - self._probe_only = probe_only - self._sync_manager = None self.control_dir = Path(tempfile.gettempdir()) / "hermes-ssh" self.control_dir.mkdir(parents=True, exist_ok=True) # Short, deterministic socket name: the path must stay under macOS's 104-byte sun_path @@ -69,16 +66,15 @@ class SSHEnvironment(BaseEnvironment): # stability across reconnects keeps ControlMaster reuse working. A probe gets its own # per-instance socket so its cleanup() can never close the agent's shared master. socket_key = f"{user}@{host}:{port}" - if self._probe_only: + if probe_only: socket_key = f"{socket_key}:probe:{self._session_id}" _socket_id = hashlib.sha256(socket_key.encode()).hexdigest()[:16] self.control_socket = self.control_dir / f"{_socket_id}.sock" _ensure_ssh_available() self._establish_connection() - if self._probe_only: - self._remote_home = "" + if probe_only: + self._sync_manager = None return - self._remote_home = self._detect_remote_home() self._ensure_remote_dirs() self._sync_manager = FileSyncManager( From 8513c5984c228bd9490ef47fad887f90d20728a3 Mon Sep 17 00:00:00 2001 From: BearHuddleston Date: Sat, 5 Sep 2026 01:16:31 -0500 Subject: [PATCH 061/276] [verified] fix(gateway): fence orphan timers and reconnect interrupt claims Re-scope #98106 onto the activity-based orphan policy from #100504. Keep timer ownership across callbacks and continuations, serialize reconnect paths against interrupt claims, and avoid recursive eager-resume locking. Leave cleanup polling armed when concurrent cold reuse is rejected. --- tests/test_tui_gateway_server.py | 13 +- tests/tui_gateway/test_ws_orphan_races.py | 159 ++++++++++++++++++++++ tui_gateway/methods_prompt.py | 10 +- tui_gateway/methods_session.py | 70 ++++++---- tui_gateway/server.py | 4 +- tui_gateway/session_lifecycle.py | 34 +++-- 6 files changed, 244 insertions(+), 46 deletions(-) create mode 100644 tests/tui_gateway/test_ws_orphan_races.py diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index ec28bc1191..07ee399e92 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -5439,7 +5439,7 @@ def test_resume_rebind_cancels_pending_ws_orphan_reap(monkeypatch): def test_claim_or_reuse_live_winner_cancels_pending_reap(monkeypatch): - """A resume that reuses the live winner cancels the winner's pending reap.""" + """The winner's pending reap is cancelled only once guarded reuse is accepted.""" cancelled = [] class _Timer: @@ -5472,6 +5472,17 @@ def test_claim_or_reuse_live_winner_cancels_pending_reap(monkeypatch): ) assert live == ("winner-sid", winner) + assert "winner-sid" in server._pending_ws_reaps + assert cancelled == [] + assert winner["transport"] is server._detached_ws_transport + + transport = object() + monkeypatch.setattr(server, "current_transport", lambda: transport) + ctx = server._Resume(1, {"omit_messages": True}, "stored-claim") + response = server._resume_reuse_live(ctx, *live) + + assert response["result"]["session_id"] == "winner-sid" + assert winner["transport"] is transport assert "winner-sid" not in server._pending_ws_reaps assert len(cancelled) == 1 finally: diff --git a/tests/tui_gateway/test_ws_orphan_races.py b/tests/tui_gateway/test_ws_orphan_races.py new file mode 100644 index 0000000000..6fd1062733 --- /dev/null +++ b/tests/tui_gateway/test_ws_orphan_races.py @@ -0,0 +1,159 @@ +"""Orphan callbacks own only their detachment, never a later reconnect.""" + +from contextlib import nullcontext +import threading +from types import SimpleNamespace +from unittest.mock import Mock + +import pytest + +from tui_gateway import server + + +@pytest.mark.parametrize("phase", ["before_callback", "before_continuation", "before_initial_timer", "cold_resume_claim"]) +def test_obsolete_orphan_cannot_replace_new_detachment(monkeypatch, phase): + timers = [] + + class Timer: + def __init__(self, delay, callback): + self.callback = callback + timers.append(self) + + def start(self): + pass + + def cancel(self): + pass # A dispatched callback can still execute after cancel(). + + sid = "generation-race" + session = dict(transport=server._detached_ws_transport, running=True, + agent=SimpleNamespace(get_activity_summary=lambda: {"seconds_since_activity": 0})) + monkeypatch.setattr(server, "_sessions", {sid: session}) + monkeypatch.setattr(server, "_pending_ws_reaps", {}) + monkeypatch.setattr(server.threading, "Timer", Timer) + monkeypatch.setattr(server, "_WS_ORPHAN_REAP_GRACE_S", 20) + monkeypatch.setattr(server, "_WS_ORPHAN_ACTIVITY_STALE_S", 600) + monkeypatch.setattr(server, "_session_has_active_delegations", lambda *a: False) + if phase == "before_initial_timer": + transport = object() + session["transport"] = transport + newest = None + + class DisconnectLock: + def __enter__(self): + pass + + def __exit__(self, *args): + # A new client reconnects and drops as the old disconnect + # releases its claim, before any out-of-lock scheduling. + nonlocal newest + server._cancel_ws_orphan_reap(sid) + server._schedule_ws_orphan_reap(sid) + newest = timers[-1] + + monkeypatch.setattr(server, "_session_resume_lock", DisconnectLock()) + assert server._close_sessions_for_transport(transport) == (0, 1) + assert server._pending_ws_reaps[sid] is newest + return + server._schedule_ws_orphan_reap(sid) + old = timers[-1] + if phase == "cold_resume_claim": + # A cold resume missed the live lookup before a concurrent resume won. + # Its claim discovers that winner while orphan interrupt I/O is in flight. + session["session_key"] = sid + monkeypatch.setattr(server, "_WS_ORPHAN_ACTIVITY_STALE_S", 0) + replies = [] + + def resume_during_interrupt(*a, **kw): + ctx = server._Resume(1, {}, sid) + replies.append(ctx.claim("unused", {})) + + monkeypatch.setattr(server, "_interrupt_session_turn", resume_during_interrupt) + old.callback() + assert replies[0]["error"]["code"] == 4009 + assert session["transport"] is server._detached_ws_transport + assert session["_client_gone_interrupt_requested"] + assert len(timers) == 2 + assert server._pending_ws_reaps[sid] is timers[-1] + return + + def redetach(): + server._cancel_ws_orphan_reap(sid) + session["transport"] = server._detached_ws_transport + server._schedule_ws_orphan_reap(sid) + return timers[-1] + + if phase == "before_callback": + newest = redetach() + else: + # Interrupt I/O runs outside the resume lock. A reconnect/redetach + # can win before the old callback registers its next poll. + monkeypatch.setattr(server, "_WS_ORPHAN_ACTIVITY_STALE_S", 0) + def interrupt(*a, **kw): + nonlocal newest + session.pop("_client_gone_interrupt_requested", None) + newest = redetach() + monkeypatch.setattr(server, "_interrupt_session_turn", interrupt) + newest = None + old.callback() + assert server._pending_ws_reaps[sid] is newest + assert timers == [old, newest] + + +@pytest.mark.parametrize("path", ["unpersisted", "reuse", "eager", "activate", "prompt"]) +@pytest.mark.parametrize("claim", ["already_claimed", "wins_lock", "retired"]) +def test_reconnect_cannot_cross_orphan_interrupt_claim(monkeypatch, path, claim): + sid = "interrupt-race" + session = dict(transport=server._detached_ws_transport, running=True, + history_lock=threading.Lock(), history=[], session_key="stored", + agent=SimpleNamespace(model="test"), queued_prompt=None) + session["_client_gone_interrupt_requested"] = claim == "already_claimed" + monkeypatch.setattr(server, "_sessions", {sid: session}) + monkeypatch.setattr(server, "_pending_ws_reaps", {sid: Mock()}) + transport = object() + monkeypatch.setattr(server, "current_transport", lambda: transport) + monkeypatch.setattr(server, "_resolve_model", lambda: "test") + monkeypatch.setattr(server, "_ensure_active_session_slot", lambda *a: None) + monkeypatch.setattr(server, "_legacy_group_fence_error", lambda *a: None) + monkeypatch.setattr(server, "_session_uses_compute_host", lambda *a: False) + monkeypatch.setattr(server, "_load_dashboard_process_isolation_config", lambda: {}) + monkeypatch.setattr(server, "_handle_busy_submit", lambda *a, **kw: {"result": {"queued": True}}) + monkeypatch.setattr(server, "_sess", lambda *a: (session, None)) + + class ResumeLock: + held = False + + def __enter__(self): + assert not self.held, "resume path recursively acquired a non-reentrant lock" + self.held = True + if claim == "wins_lock": + session["_client_gone_interrupt_requested"] = True + elif claim == "retired": + server._sessions.pop(sid, None) + + def __exit__(self, *args): + self.held = False + + monkeypatch.setattr(server, "_session_resume_lock", ResumeLock()) + ctx = SimpleNamespace(rid=1, owns_db=False, db=None, cols=80, omit_messages=True, + defer_history=False, target="stored", profile=None, + profile_home=None, profile_resume_cwd=None, found={}, + messages=lambda history: [], mint=lambda: ("unused", "tui", "."), + restore=lambda: ([], [], []), display_prefix=lambda: []) + if path == "eager": + monkeypatch.setattr(server, "_profile_build_scope", lambda *a: nullcontext()) + monkeypatch.setattr(server, "_make_agent_in_context", lambda *a, **kw: Mock()) + monkeypatch.setattr(server, "_find_live_session_by_key", lambda *a: (sid, session)) + response = server._resume_eager(ctx) + elif path == "unpersisted": + response = server._resume_live_unpersisted(ctx, sid, session) + elif path == "reuse": + response = server._resume_reuse_live(ctx, sid, session) + else: + name = {"activate": "session.activate", "prompt": "prompt.submit"}[path] + response = server.handle_request({"jsonrpc": "2.0", "id": 1, "method": name, + "params": {"session_id": sid, "text": "continue", "omit_messages": True}}) + assert response.get("error", {}).get("code") == (4007 if claim == "retired" else 4009) + assert session["transport"] is server._detached_ws_transport + assert sid in server._pending_ws_reaps + assert session["queued_prompt"] is None diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index b37588e7b0..84b76b13d7 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -574,8 +574,14 @@ def _(rid, params: dict) -> dict: return _err(rid, 4121, "hosted room turns do not support isolated compute workers yet") # Re-bind to the current transport: streaming must stay on the active websocket even # if a disconnect/fallback moved the session to stdio. - if (t := current_transport()) is not None: - session["transport"] = t + with _session_resume_lock: + if _sessions.get(sid) is not session: + return _err(rid, 4007, "session no longer live; retry resume") + if session.get("_client_gone_interrupt_requested"): + return _err(rid, 4009, "session disconnect interrupt settling") + if (t := current_transport()) is not None: + session["transport"] = t + _cancel_ws_orphan_reap(sid) # Claim the turn against a possibly-running session (busy/queued reply, else fall # through once ``running`` is observed False). The provider interrupt happens after # history_lock is released (a non-interruptible tool may hold it); if the old turn diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 1c2e13efdf..424266827b 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -523,16 +523,17 @@ def _resume_live_unpersisted(ctx: _Resume, live_sid: str, live: dict) -> dict: sentinel-parked the record) or it fires against this client.""" if ctx.owns_db: _release_db(ctx.db) - live["last_active"] = time.time() - if (transport := current_transport()) is not None: - # This resume reattaches the live record. A lazy session (no state.db row yet — every fresh Bot - # Chat) that was sentinel-parked by a WS drop MUST be rebound here, or it keeps the drop sentinel - # and the armed orphan-reap Timer fires against a client that is attached right now — the - # unpersisted sibling of the storm-killer paths (#91276). - with live.setdefault("history_lock", threading.Lock()): - live["transport"] = transport - live.setdefault("viewers", {})[transport] = time.time() - _cancel_ws_orphan_reap(live_sid) + with _session_resume_lock: + if _sessions.get(live_sid) is not live: + return _err(ctx.rid, 4007, "session no longer live; retry resume") + if live.get("_client_gone_interrupt_requested"): + return _err(ctx.rid, 4009, "session disconnect interrupt settling") + live["last_active"] = time.time() + if (transport := current_transport()) is not None: + with live.setdefault("history_lock", threading.Lock()): + live["transport"] = transport + live.setdefault("viewers", {})[transport] = time.time() + _cancel_ws_orphan_reap(live_sid) history = live.get("history") or [] return _ok(ctx.rid, _attach_todo_state({ "session_id": live_sid, "stored_session_id": str(live.get("session_key") or ""), @@ -631,21 +632,26 @@ def _resume_reuse_live(ctx: _Resume, sid: str, session: dict) -> dict: """Reattach an already-live session under the resume lock (held across the client-gone check, transport rebind and reap cancel so grace expiry is atomic).""" with _session_resume_lock: - if _sessions.get(sid) is not session: - return _err(ctx.rid, 4007, "session no longer live; retry resume") - if session.get("_client_gone_interrupt_requested"): - return _err(ctx.rid, 4009, "session disconnect interrupt settling") - _cancel_ws_orphan_reap(sid) # unconditionally: the fast path must never race the reap Timer - payload = _live_session_payload(sid, session, cols=ctx.cols, touch=True, omit_messages=ctx.omit_messages, - transport=current_transport() or _stdio_transport) - payload["resumed"] = ctx.target - if ctx.defer_history: - payload.update(messages=[], hydrating=bool(session.get("resume_hydrating")), - message_count=int(session.get("resume_message_count") or payload["message_count"])) - # A lazy watch session never owns a run loop — overlay the child-run registry. - if session.get("agent") is None and _child_run_active(ctx.target): - payload.update(running=True, status="streaming") - return _ok(ctx.rid, payload) + return _resume_reuse_live_locked(ctx, sid, session) + + +def _resume_reuse_live_locked(ctx: _Resume, sid: str, session: dict) -> dict: + """Reuse with _session_resume_lock already held (including the eager double-check).""" + if _sessions.get(sid) is not session: + return _err(ctx.rid, 4007, "session no longer live; retry resume") + if session.get("_client_gone_interrupt_requested"): + return _err(ctx.rid, 4009, "session disconnect interrupt settling") + _cancel_ws_orphan_reap(sid) # unconditionally: the fast path must never race the reap Timer + payload = _live_session_payload(sid, session, cols=ctx.cols, touch=True, omit_messages=ctx.omit_messages, + transport=current_transport() or _stdio_transport) + payload["resumed"] = ctx.target + if ctx.defer_history: + payload.update(messages=[], hydrating=bool(session.get("resume_hydrating")), + message_count=int(session.get("resume_message_count") or payload["message_count"])) + # A lazy watch session never owns a run loop — overlay the child-run registry. + if session.get("agent") is None and _child_run_active(ctx.target): + payload.update(running=True, status="streaming") + return _ok(ctx.rid, payload) def _resume_response( @@ -751,7 +757,7 @@ def _resume_eager(ctx: _Resume) -> dict: if live is not None: with contextlib.suppress(Exception): agent.close() - return _resume_reuse_live(ctx, *live) + return _resume_reuse_live_locked(ctx, *live) try: with _profile_build_scope(ctx.profile_home): _init_session(sid, ctx.target, agent, history, cols=ctx.cols, cwd=ctx.profile_resume_cwd, @@ -890,9 +896,15 @@ def _(rid, params: dict) -> dict: @_session_method("session.activate") def _(rid, params: dict, session: dict) -> dict: """Attach the frontend to a live TUI session without closing the previously focused one.""" - return _ok(rid, _live_session_payload( - str(params.get("session_id") or ""), session, touch=True, transport=current_transport() or _stdio_transport, - omit_messages=is_truthy_value(params.get("omit_messages", False)))) + sid = str(params.get("session_id") or "") + with _session_resume_lock: + if _sessions.get(sid) is not session: + return _err(rid, 4007, "session no longer live; retry resume") + if session.get("_client_gone_interrupt_requested"): + return _err(rid, 4009, "session disconnect interrupt settling") + return _ok(rid, _live_session_payload( + sid, session, touch=True, transport=current_transport() or _stdio_transport, + omit_messages=is_truthy_value(params.get("omit_messages", False)))) @method("session.delete") diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 49266de5ef..93c8a0a0b3 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -2444,8 +2444,8 @@ def _claim_or_reuse_live(sid: str, session_key: str, record: dict, lease) -> tup if live is not None: if lease is not None: lease.release() - # The winner is being reattached: a pending ws-orphan reap must not fire against the reclaimed client. - _cancel_ws_orphan_reap(live[0]) + # Guarded reuse cancels the reap only after accepting reattachment; + # a rejected resume must leave an in-flight orphan interrupt polling. return live with _sessions_lock: _sessions[sid] = record diff --git a/tui_gateway/session_lifecycle.py b/tui_gateway/session_lifecycle.py index c27b6e5419..3fc43e51f9 100644 --- a/tui_gateway/session_lifecycle.py +++ b/tui_gateway/session_lifecycle.py @@ -463,7 +463,9 @@ def _ws_orphan_turn_activity_is_fresh(session: dict) -> bool: return False -def _schedule_ws_orphan_reap(sid: str, *, delay_s: float | None = None) -> None: +def _schedule_ws_orphan_reap( + sid: str, *, delay_s: float | None = None, _expected_timer: threading.Timer | None = None, +) -> None: """After a grace window, reap session ``sid`` iff it's still orphaned. Called from the WS-disconnect path; a reconnect or ``session.resume`` cancels the reap by re-binding a live transport. Disabled when grace is 0.""" if _WS_ORPHAN_REAP_GRACE_S <= 0: @@ -473,13 +475,14 @@ def _schedule_ws_orphan_reap(sid: str, *, delay_s: float | None = None) -> None: # Serialize the re-check against session.resume (rebinds under _session_resume_lock). Claim teardown by popping # under both locks, then release the resume lock before slow finalization. Order: resume_lock -> sessions_lock. reschedule_delay = interrupt_session = session = None - with _session_resume_lock: - # Drop this Timer's registration so a concurrent _cancel_ws_orphan_reap can't cancel a dead Timer while a - # rescheduled one (registered below) is the owner. - with _sessions_lock: - _pending_ws_reaps.pop(sid, None) + with _session_resume_lock, _sessions_lock: + # Keep ownership through interrupt I/O and continuation registration. A cancelled + # callback may already be dispatched, but cannot act on a later detachment. + if _pending_ws_reaps.get(sid) is not timer: + return current = _sessions.get(sid) if current is None or not _ws_session_is_detached(current): + _pending_ws_reaps.pop(sid, None) return if _session_has_active_delegations(sid, current): reschedule_delay = _WS_ORPHAN_REAP_GRACE_S @@ -506,6 +509,8 @@ def _schedule_ws_orphan_reap(sid: str, *, delay_s: float | None = None) -> None: current["_client_gone_interrupt_requested"] = True interrupt_session = current reschedule_delay = _WS_ORPHAN_INTERRUPT_REAP_POLL_S + if reschedule_delay is None: + _pending_ws_reaps.pop(sid, None) if interrupt_session is not None: try: isolated = _interrupt_session_turn(sid, interrupt_session, request_id=f"client-gone-{sid}") @@ -513,18 +518,21 @@ def _schedule_ws_orphan_reap(sid: str, *, delay_s: float | None = None) -> None: except Exception: logger.exception("client_gone interrupt failed sid=%s", sid) with _sessions_lock: - if _sessions.get(sid) is interrupt_session: + if (_sessions.get(sid) is interrupt_session + and _pending_ws_reaps.get(sid) is timer): interrupt_session.pop("_client_gone_interrupt_requested", None) if reschedule_delay is not None: - _schedule_ws_orphan_reap(sid, delay_s=reschedule_delay) + _schedule_ws_orphan_reap(sid, delay_s=reschedule_delay, _expected_timer=timer) return if session is not None and session.get("_client_gone_interrupt_requested"): logger.info("client_gone sid=%s action=reap", sid) _teardown_popped_session(session, end_reason="ws_orphan_reap") - timer = threading.Timer(_WS_ORPHAN_REAP_GRACE_S if delay_s is None else max(0.0, delay_s), _reap) - timer.daemon = True with _sessions_lock: + if _expected_timer is not None and _pending_ws_reaps.get(sid) is not _expected_timer: + return + timer = threading.Timer(_WS_ORPHAN_REAP_GRACE_S if delay_s is None else max(0.0, delay_s), _reap) + timer.daemon = True prior = _pending_ws_reaps.pop(sid, None) _pending_ws_reaps[sid] = timer if prior is not None: @@ -569,12 +577,14 @@ def _close_sessions_for_transport(transport, *, end_reason: str = "ws_disconnect current["transport"] = _detached_ws_transport current.pop("_client_gone_interrupt_requested", None) should_schedule_reap = True + # Register before releasing the detachment claim: an old disconnect + # must not arm its first timer over a reconnect's newer detachment. + with contextlib.suppress(Exception): + _schedule_ws_orphan_reap(sid) if claimed_for_teardown is not None: reaped += _teardown_popped_session(claimed_for_teardown, end_reason=end_reason) elif should_schedule_reap: detached += 1 - with contextlib.suppress(Exception): - _schedule_ws_orphan_reap(sid) return reaped, detached From f2077a020989329b731f45208c8cb3e4b7bdf221 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 13:38:55 +0530 Subject: [PATCH 062/276] refactor(tui-gateway): one _reattach_refusal helper for the live-identity/interrupt-settling guard The 4007/4009 guard the salvaged fix added to prompt.submit, session.activate, _resume_live_unpersisted and _resume_reuse_live_locked was the same six lines four times. One helper in session_lifecycle.py (next to _cancel_ws_orphan_reap, whose contract it mirrors) so the next reattach path cannot drift from the others. Behaviour unchanged; the 19 race tests cover all four call sites. --- tui_gateway/methods_prompt.py | 6 ++---- tui_gateway/methods_session.py | 18 ++++++------------ tui_gateway/session_lifecycle.py | 11 +++++++++++ 3 files changed, 19 insertions(+), 16 deletions(-) diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index 84b76b13d7..4455fa22c1 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -575,10 +575,8 @@ def _(rid, params: dict) -> dict: # Re-bind to the current transport: streaming must stay on the active websocket even # if a disconnect/fallback moved the session to stdio. with _session_resume_lock: - if _sessions.get(sid) is not session: - return _err(rid, 4007, "session no longer live; retry resume") - if session.get("_client_gone_interrupt_requested"): - return _err(rid, 4009, "session disconnect interrupt settling") + if (refusal := _reattach_refusal(rid, sid, session)) is not None: + return refusal if (t := current_transport()) is not None: session["transport"] = t _cancel_ws_orphan_reap(sid) diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 424266827b..9190987f88 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -524,10 +524,8 @@ def _resume_live_unpersisted(ctx: _Resume, live_sid: str, live: dict) -> dict: if ctx.owns_db: _release_db(ctx.db) with _session_resume_lock: - if _sessions.get(live_sid) is not live: - return _err(ctx.rid, 4007, "session no longer live; retry resume") - if live.get("_client_gone_interrupt_requested"): - return _err(ctx.rid, 4009, "session disconnect interrupt settling") + if (refusal := _reattach_refusal(ctx.rid, live_sid, live)) is not None: + return refusal live["last_active"] = time.time() if (transport := current_transport()) is not None: with live.setdefault("history_lock", threading.Lock()): @@ -637,10 +635,8 @@ def _resume_reuse_live(ctx: _Resume, sid: str, session: dict) -> dict: def _resume_reuse_live_locked(ctx: _Resume, sid: str, session: dict) -> dict: """Reuse with _session_resume_lock already held (including the eager double-check).""" - if _sessions.get(sid) is not session: - return _err(ctx.rid, 4007, "session no longer live; retry resume") - if session.get("_client_gone_interrupt_requested"): - return _err(ctx.rid, 4009, "session disconnect interrupt settling") + if (refusal := _reattach_refusal(ctx.rid, sid, session)) is not None: + return refusal _cancel_ws_orphan_reap(sid) # unconditionally: the fast path must never race the reap Timer payload = _live_session_payload(sid, session, cols=ctx.cols, touch=True, omit_messages=ctx.omit_messages, transport=current_transport() or _stdio_transport) @@ -898,10 +894,8 @@ def _(rid, params: dict, session: dict) -> dict: """Attach the frontend to a live TUI session without closing the previously focused one.""" sid = str(params.get("session_id") or "") with _session_resume_lock: - if _sessions.get(sid) is not session: - return _err(rid, 4007, "session no longer live; retry resume") - if session.get("_client_gone_interrupt_requested"): - return _err(rid, 4009, "session disconnect interrupt settling") + if (refusal := _reattach_refusal(rid, sid, session)) is not None: + return refusal return _ok(rid, _live_session_payload( sid, session, touch=True, transport=current_transport() or _stdio_transport, omit_messages=is_truthy_value(params.get("omit_messages", False)))) diff --git a/tui_gateway/session_lifecycle.py b/tui_gateway/session_lifecycle.py index 3fc43e51f9..5116186764 100644 --- a/tui_gateway/session_lifecycle.py +++ b/tui_gateway/session_lifecycle.py @@ -443,6 +443,17 @@ def _cancel_ws_orphan_reap(sid: str) -> None: timer.cancel() +def _reattach_refusal(rid, sid: str, session: dict) -> dict | None: + """Under ``_session_resume_lock``: the error a reattaching RPC (resume/activate/prompt.submit) must return + instead of rebinding — ``session`` is no longer the live record for ``sid``, or a client-gone interrupt is + still settling and the reap Timer must keep polling. None when the reattach may proceed.""" + if _sessions.get(sid) is not session: + return _err(rid, 4007, "session no longer live; retry resume") + if session.get("_client_gone_interrupt_requested"): + return _err(rid, 4009, "session disconnect interrupt settling") + return None + + def _ws_orphan_turn_activity_is_fresh(session: dict) -> bool: """Whether a detached RUNNING turn's activity clock (``_touch_activity``) is still fresh — the reaper must NOT interrupt healthy detached work (closed laptop). Conservative: disabled threshold, missing/opaque agent, unreadable From c6cb7111a24f787d6cb542c3bbcad7a394ade50c Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 13:40:41 +0530 Subject: [PATCH 063/276] perf(tui-gateway): session.activate builds its payload outside _session_resume_lock The salvaged fix correctly put the activate guard + transport rebind under the process-wide resume lock, but it dragged _live_session_payload in with them. The Desktop passes omit_messages=true (cheap), the Ink TUI does not: every TUI session switch then read the full persisted history from the profile DB while holding the lock that serializes every resume, disconnect and reap Timer. Extract _rebind_live_transport from _live_session_payload; activate does guard + rebind under the lock (the part that must be atomic with grace expiry) and builds the payload after releasing it. --- tui_gateway/methods_session.py | 9 ++++++--- tui_gateway/server.py | 19 ++++++++++++------- 2 files changed, 18 insertions(+), 10 deletions(-) diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 9190987f88..dfc21ff6f2 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -893,12 +893,15 @@ def _(rid, params: dict) -> dict: def _(rid, params: dict, session: dict) -> dict: """Attach the frontend to a live TUI session without closing the previously focused one.""" sid = str(params.get("session_id") or "") + # Only the rebind is atomic with grace expiry; the payload (a DB history read unless + # ``omit_messages``) must not hold the process-wide resume lock. with _session_resume_lock: if (refusal := _reattach_refusal(rid, sid, session)) is not None: return refusal - return _ok(rid, _live_session_payload( - sid, session, touch=True, transport=current_transport() or _stdio_transport, - omit_messages=is_truthy_value(params.get("omit_messages", False)))) + with session["history_lock"]: + _rebind_live_transport(sid, session, current_transport() or _stdio_transport) + return _ok(rid, _live_session_payload( + sid, session, touch=True, omit_messages=is_truthy_value(params.get("omit_messages", False)))) @method("session.delete") diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 93c8a0a0b3..82e3c20d6f 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -2686,6 +2686,17 @@ def _live_visible_history(session: dict, db, in_memory_fallback: list[dict]) -> return in_memory_fallback +def _rebind_live_transport(sid: str, session: dict, transport: Transport) -> None: + """Point a live session at ``transport`` (caller holds ``history_lock``).""" + session["transport"] = transport + # Every transport that showed this session (pop-outs resume the same sid); on disconnect the last + # viewer becomes the transport instead of the drop sentinel. + session.setdefault("viewers", {})[transport] = time.time() + # See #83716. + if transport is not _detached_ws_transport: + _cancel_ws_orphan_reap(sid) # the client is back — a pending ws-orphan reap must not fire + + def _live_session_payload( sid: str, session: dict, *, cols: int | None = None, touch: bool = False, transport: Transport | None = None, omit_messages: bool = False) -> dict: @@ -2693,13 +2704,7 @@ def _live_session_payload( if cols is not None: session["cols"] = cols if transport is not None: - session["transport"] = transport - # Every transport that showed this session (pop-outs resume the same sid); on disconnect the last - # viewer becomes the transport instead of the drop sentinel. - session.setdefault("viewers", {})[transport] = time.time() - # See #83716. - if transport is not _detached_ws_transport: - _cancel_ws_orphan_reap(sid) # the client is back — a pending ws-orphan reap must not fire + _rebind_live_transport(sid, session, transport) if touch: # #84417: do not re-fire the live turn's original user text from a stale server-queue # self-duplicate after settle. From 212ed99d9668b5356acd7f309c8ac0e708772000 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:00:08 +0530 Subject: [PATCH 064/276] refactor(tui-gateway): lazy resume reattaches through _rebind_live_transport _resume_live_unpersisted hand-rolled the same transport + viewers + reap-cancel sequence _rebind_live_transport now owns. Route it through the helper; the stdio case (no current transport) keeps cancelling the reap as before. --- tui_gateway/methods_session.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index dfc21ff6f2..078f321769 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -529,9 +529,9 @@ def _resume_live_unpersisted(ctx: _Resume, live_sid: str, live: dict) -> dict: live["last_active"] = time.time() if (transport := current_transport()) is not None: with live.setdefault("history_lock", threading.Lock()): - live["transport"] = transport - live.setdefault("viewers", {})[transport] = time.time() - _cancel_ws_orphan_reap(live_sid) + _rebind_live_transport(live_sid, live, transport) + else: + _cancel_ws_orphan_reap(live_sid) history = live.get("history") or [] return _ok(ctx.rid, _attach_todo_state({ "session_id": live_sid, "stored_session_id": str(live.get("session_key") or ""), From 01ae7a5668ce0fa2efca524a4567cacdd0786c95 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:06:12 +0530 Subject: [PATCH 065/276] refactor(tui-gateway): _rebind_live_transport lives with the reattach contract in session_lifecycle It is lifecycle logic (detached sentinel, reap cancel) and its siblings _reattach_refusal / _cancel_ws_orphan_reap are already there; server.py is the facade and gets no new definitions. Also trims the _claim_or_reuse_live comment to point at where the reap cancel now happens. --- tui_gateway/server.py | 15 ++------------- tui_gateway/session_lifecycle.py | 17 ++++++++++++++--- 2 files changed, 16 insertions(+), 16 deletions(-) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 82e3c20d6f..671def454a 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -2444,8 +2444,8 @@ def _claim_or_reuse_live(sid: str, session_key: str, record: dict, lease) -> tup if live is not None: if lease is not None: lease.release() - # Guarded reuse cancels the reap only after accepting reattachment; - # a rejected resume must leave an in-flight orphan interrupt polling. + # The reap is cancelled by the guarded reuse (_reattach_refusal), not here: a rejected + # reattach must leave an in-flight orphan interrupt polling. return live with _sessions_lock: _sessions[sid] = record @@ -2686,17 +2686,6 @@ def _live_visible_history(session: dict, db, in_memory_fallback: list[dict]) -> return in_memory_fallback -def _rebind_live_transport(sid: str, session: dict, transport: Transport) -> None: - """Point a live session at ``transport`` (caller holds ``history_lock``).""" - session["transport"] = transport - # Every transport that showed this session (pop-outs resume the same sid); on disconnect the last - # viewer becomes the transport instead of the drop sentinel. - session.setdefault("viewers", {})[transport] = time.time() - # See #83716. - if transport is not _detached_ws_transport: - _cancel_ws_orphan_reap(sid) # the client is back — a pending ws-orphan reap must not fire - - def _live_session_payload( sid: str, session: dict, *, cols: int | None = None, touch: bool = False, transport: Transport | None = None, omit_messages: bool = False) -> dict: diff --git a/tui_gateway/session_lifecycle.py b/tui_gateway/session_lifecycle.py index 5116186764..e7a77fcb37 100644 --- a/tui_gateway/session_lifecycle.py +++ b/tui_gateway/session_lifecycle.py @@ -444,9 +444,9 @@ def _cancel_ws_orphan_reap(sid: str) -> None: def _reattach_refusal(rid, sid: str, session: dict) -> dict | None: - """Under ``_session_resume_lock``: the error a reattaching RPC (resume/activate/prompt.submit) must return - instead of rebinding — ``session`` is no longer the live record for ``sid``, or a client-gone interrupt is - still settling and the reap Timer must keep polling. None when the reattach may proceed.""" + """Under ``_session_resume_lock``: why a reattaching RPC (resume/activate/prompt.submit) must NOT rebind + ``session`` — it is stale, or a client-gone interrupt is still settling and the reap Timer must keep + polling. None when the reattach may proceed.""" if _sessions.get(sid) is not session: return _err(rid, 4007, "session no longer live; retry resume") if session.get("_client_gone_interrupt_requested"): @@ -454,6 +454,17 @@ def _reattach_refusal(rid, sid: str, session: dict) -> dict | None: return None +def _rebind_live_transport(sid: str, session: dict, transport: Transport) -> None: + """Point a live session at ``transport`` (caller holds ``history_lock``).""" + session["transport"] = transport + # Every transport that showed this session (pop-outs resume the same sid); on disconnect the last + # viewer becomes the transport instead of the drop sentinel. + session.setdefault("viewers", {})[transport] = time.time() + # See #83716. + if transport is not _detached_ws_transport: + _cancel_ws_orphan_reap(sid) # the client is back — a pending ws-orphan reap must not fire + + def _ws_orphan_turn_activity_is_fresh(session: dict) -> bool: """Whether a detached RUNNING turn's activity clock (``_touch_activity``) is still fresh — the reaper must NOT interrupt healthy detached work (closed laptop). Conservative: disabled threshold, missing/opaque agent, unreadable From e3b6e4050195634224ec1b6bdccc577375d827a4 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 13:27:11 +0530 Subject: [PATCH 066/276] chore(contributors): map bounce12340 for the #102496 salvage --- .../emails/128559392+bounce12340@users.noreply.github.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/128559392+bounce12340@users.noreply.github.com diff --git a/contributors/emails/128559392+bounce12340@users.noreply.github.com b/contributors/emails/128559392+bounce12340@users.noreply.github.com new file mode 100644 index 0000000000..cebf5875b5 --- /dev/null +++ b/contributors/emails/128559392+bounce12340@users.noreply.github.com @@ -0,0 +1,2 @@ +bounce12340 +# PR #102496 salvage From 4f1fbc47cde519a343a2afb78eae866490ea2fb9 Mon Sep 17 00:00:00 2001 From: Josh Tsai <128559392+bounce12340@users.noreply.github.com> Date: Thu, 3 Sep 2026 21:45:52 +0000 Subject: [PATCH 067/276] fix(desktop): let foreground bot opens preempt spawn-cap hydration LocalBackendSpawnCoordinator is FIFO with maxBackends=3, so launch hydration queues ~27 ensureBackend calls and a user click times out waiting for a slot. Reserve a foreground slot, drain foreground first, and fail background slot-wait quietly. Fixes #102281. --- .../electron/backend-dial-claim.test.ts | 6 +- apps/desktop/electron/main.ts | 138 ++++++++++++--- .../electron/pool-spawn-coordinator.test.ts | 160 ++++++++++++++++- .../electron/pool-spawn-coordinator.ts | 167 ++++++++++++++++-- apps/desktop/electron/preload.ts | 2 +- .../hooks/use-session-actions/index.ts | 6 +- apps/desktop/src/global.d.ts | 6 +- apps/desktop/src/sdk/index.ts | 7 +- apps/desktop/src/sdk/profile-routing.test.ts | 6 +- apps/desktop/src/store/gateway.ts | 99 ++++++++--- apps/desktop/src/store/profile.ts | 5 +- 11 files changed, 525 insertions(+), 77 deletions(-) diff --git a/apps/desktop/electron/backend-dial-claim.test.ts b/apps/desktop/electron/backend-dial-claim.test.ts index 6aa7ce1353..cb3d6099ec 100644 --- a/apps/desktop/electron/backend-dial-claim.test.ts +++ b/apps/desktop/electron/backend-dial-claim.test.ts @@ -126,10 +126,10 @@ describe('main.ts wiring for #90812', () => { it('routes the profile-scoped dial IPC through the single-owner claim', () => { const handlerStart = mainSource.indexOf("ipcMain.handle('hermes:connection', ") expect(handlerStart).toBeGreaterThan(-1) - const body = mainSource.slice(handlerStart, handlerStart + 900) + const body = mainSource.slice(handlerStart, handlerStart + 1200) expect(body).toContain('backendDialClaims.run(') - expect(body).toContain('ensureBackend(profile)') + expect(body).toContain('ensureBackend(profile, { spawnPriority })') }) it('routes the registry-scoped dial IPC through the claim keyed by backendScopeKey(connectionId, profile)', () => { @@ -138,7 +138,7 @@ describe('main.ts wiring for #90812', () => { const body = mainSource.slice(handlerStart, handlerStart + 1_200) expect(body).toContain('backendDialClaims.run(backendScopeKey(id, profile)') - expect(body).toContain('ensureRegistryBackend(id, profile)') + expect(body).toContain("ensureRegistryBackend(id, profile, '', { spawnPriority })") }) // The four IPC/probe surfaces below call ensureRegistryBackend()/ensureBackend() diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 5116cdb7bc..40df34ff80 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -287,7 +287,9 @@ import { import { selectPoolEvictions } from './pool-eviction' import { clampPoolLimits, parsePoolLimits, POOL_LIMITS_DEFAULTS } from './pool-limits' import { + isBackgroundSlotWaitTimeout, LocalBackendSpawnCoordinator, + type LocalBackendSpawnPriority, type LocalBackendSpawnRequest, releaseLocalBackendSlotAfterExit } from './pool-spawn-coordinator' @@ -1492,6 +1494,46 @@ const localBackendSpawnCoordinator = new LocalBackendSpawnCoordinator(poolLimits // the queued ticket fails before the renderer does and the user sees why. const POOL_SLOT_WAIT_MS = 30_000 +function spawnPriorityFrom(value): LocalBackendSpawnPriority { + return value === 'foreground' ? 'foreground' : 'background' +} + +const pendingForegroundSpawns = new Set() + +function markForegroundSpawn(poolKey): void { + if (poolKey) { + pendingForegroundSpawns.add(String(poolKey)) + } +} + +function takeForegroundSpawn(poolKey): boolean { + const key = String(poolKey || '') + + if (!key || !pendingForegroundSpawns.has(key)) { + return false + } + + pendingForegroundSpawns.delete(key) + + return true +} + +function promoteInFlightLocalSpawn(poolKey, spawnPriority: LocalBackendSpawnPriority): void { + if (spawnPriority !== 'foreground') { + return + } + + markForegroundSpawn(poolKey) + const existing = backendPool.get(poolKey) + + if (!existing) { + return + } + + existing.spawnPriority = 'foreground' + existing.localBackendSpawnRequest?.promote?.('foreground') +} + function poolMaxBackends() { return poolLimits.maxBackends } @@ -11382,8 +11424,9 @@ function profileRouteOptions(profile, request?) { // Resolve a backend connection for the given profile, per the routing table in // resolveProfileBackendRoute(). An empty / unknown profile resolves to the // primary, so legacy callers are unchanged. -async function ensureBackend(profile) { +async function ensureBackend(profile, opts: { spawnPriority?: LocalBackendSpawnPriority } = {}) { const key = profile && String(profile).trim() ? String(profile).trim() : primaryProfileKey() + const spawnPriority = spawnPriorityFrom(opts.spawnPriority) profileDeletionGate.assertCanStart(key) @@ -11415,6 +11458,12 @@ async function ensureBackend(profile) { if (existing) { existing.lastActiveAt = Date.now() + + if (spawnPriority === 'foreground') { + existing.spawnPriority = 'foreground' + existing.localBackendSpawnRequest?.promote?.('foreground') + } + const connection = await existing.connectionPromise setWslBridgeProfileState(key, connection.mode !== 'remote') @@ -11432,16 +11481,23 @@ async function ensureBackend(profile) { remoteBaseUrl: null, releaseLocalBackendSlot: null, localBackendSlotKey: null, - localBackendSpawnRequest: null + localBackendSpawnRequest: null, + spawnPriority } - entry.connectionPromise = spawnPoolBackend(key, entry).catch(async error => { + entry.connectionPromise = spawnPoolBackend(key, entry, { spawnPriority }).catch(async error => { // Land the failure in desktop.log: without this a spawn that dies before // its child exists (guard rejection, runtime resolution) leaves no trace // beyond renderer-side rejections users never see in a bundle. - rememberLog( - `Hermes backend for profile "${key}" failed to start: ${error instanceof Error ? error.message : String(error)}` - ) + if (isBackgroundSlotWaitTimeout(error)) { + rememberLog( + `Profile backend "${key}" slot wait timed out (background); will retry on the next hydration` + ) + } else { + rememberLog( + `Hermes backend for profile "${key}" failed to start: ${error instanceof Error ? error.message : String(error)}` + ) + } await teardownFailedLocalBackend(key, entry) throw error @@ -11462,7 +11518,13 @@ async function ensureBackend(profile) { // a genuinely-local child when the v1 mode says remote; non-local connections // pool under the composite key from backendScopeKey() and reuse the same pool // entry lifecycle (LRU, idle reaper, touch) as per-profile local backends. -async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrelation = '') { +async function ensureRegistryBackend( + connectionId, + profile, + managedUpdateCorrelation = '', + opts: { spawnPriority?: LocalBackendSpawnPriority } = {} +) { + const spawnPriority = spawnPriorityFrom(opts.spawnPriority) const registry = readDesktopConnectionsRegistry() const id = String(connectionId || '').trim() || registry.primary const source = registry.connections.find(c => c.id === id) @@ -11519,7 +11581,7 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela const primary = await reuseMatchingPrimarySshBackend({ connectionId: id, effectiveFingerprint: resolveRegistryEffectiveFingerprint, - ensurePrimary: () => ensureBackend(profile), + ensurePrimary: () => ensureBackend(profile, { spawnPriority }), profile, registry, source @@ -11570,7 +11632,7 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela }) if (localRoute.delegate) { - return ensureBackend(profile) + return ensureBackend(profile, { spawnPriority }) } const stoppingLocal = poolStopper.inFlight(localRoute.poolKey) @@ -11584,6 +11646,11 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela if (existingLocal) { existingLocal.lastActiveAt = Date.now() + if (spawnPriority === 'foreground') { + existingLocal.spawnPriority = 'foreground' + existingLocal.localBackendSpawnRequest?.promote?.('foreground') + } + return existingLocal.connectionPromise } @@ -11598,18 +11665,26 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela remoteBaseUrl: null, releaseLocalBackendSlot: null, localBackendSlotKey: null, - localBackendSpawnRequest: null + localBackendSpawnRequest: null, + spawnPriority } localEntry.connectionPromise = spawnPoolBackend(profileKey, localEntry, { forceLocal: true, - poolKey: localRoute.poolKey + poolKey: localRoute.poolKey, + spawnPriority }).catch(async error => { // Same trace rule as the v1 pool path: a forced-local child whose spawn // rejects before the child exists must still land in desktop.log. - rememberLog( - `Hermes backend for profile "${profileKey}" (forced-local) failed to start: ${error instanceof Error ? error.message : String(error)}` - ) + if (isBackgroundSlotWaitTimeout(error)) { + rememberLog( + `Profile backend "${profileKey}" (forced-local) slot wait timed out (background); will retry on the next hydration` + ) + } else { + rememberLog( + `Hermes backend for profile "${profileKey}" (forced-local) failed to start: ${error instanceof Error ? error.message : String(error)}` + ) + } await teardownFailedLocalBackend(localRoute.poolKey, localEntry) throw error @@ -12412,7 +12487,11 @@ function teardownFailedLocalBackend(poolKey: string, entry: any): Promise // entry means THIS machine regardless of the v1 routing table); `opts.poolKey` // is the backendPool key when it differs from the profile name (composite // registry scopes) so the exit/error cleanup evicts the right entry. -async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; poolKey?: string } = {}) { +async function spawnPoolBackend( + profile, + entry, + opts: { forceLocal?: boolean; poolKey?: string; spawnPriority?: LocalBackendSpawnPriority } = {} +) { const poolKey = opts.poolKey || profile await reapOrphanedBackendsOnce() @@ -12447,11 +12526,21 @@ async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; po // pool-idle window (10 min) would hold the pool key hostage and every // later click on the profile would join that stale wait. Failing here // surfaces the "all N slots busy" reason instead of a generic boot timeout. - const spawnRequest = localBackendSpawnCoordinator.request(poolKey, { timeoutMs: POOL_SLOT_WAIT_MS }) + const markedKey = takeForegroundSpawn(poolKey) + const markedProfile = takeForegroundSpawn(profile) + const spawnPriority = + entry.spawnPriority === 'foreground' || + opts.spawnPriority === 'foreground' || + markedKey || + markedProfile + ? 'foreground' + : 'background' + entry.spawnPriority = spawnPriority + const spawnRequest = localBackendSpawnCoordinator.request(poolKey, { timeoutMs: POOL_SLOT_WAIT_MS, priority: spawnPriority }) entry.localBackendSlotKey = poolKey entry.localBackendSpawnRequest = spawnRequest - if (localBackendSpawnCoordinator.activeCount >= poolMaxBackends()) { + if (localBackendSpawnCoordinator.queuedCount > 0) { rememberLog( `Profile backend "${profile}" waiting for a free local slot (${localBackendSpawnCoordinator.activeCount}/${poolMaxBackends()} busy, ${localBackendSpawnCoordinator.queuedCount} queued)` ) @@ -14711,13 +14800,17 @@ function createWindow() { }) } -ipcMain.handle('hermes:connection', async (_event, profile) => { +ipcMain.handle('hermes:connection', async (_event, profile, extra) => { // Coalesce concurrent renderer dials for one profile scope (#90812): the // renderer-side reconnect lock is per-window, so two windows waking at once // both land here. The claim key mirrors ensureBackend()'s own profile // normalization so every spelling of the primary coalesces onto one dial. const profileKey = profile && String(profile).trim() ? String(profile).trim() : primaryProfileKey() - const connection = await backendDialClaims.run(backendScopeKey(null, profileKey), () => ensureBackend(profile)) + const spawnPriority = spawnPriorityFrom(extra && typeof extra === 'object' ? extra.priority : undefined) + // A user click may join an in-flight hydration claim; promote the queued + // slot wait before coalescing so it can take the reserved foreground slot. + promoteInFlightLocalSpawn(profileKey, spawnPriority) + const connection = await backendDialClaims.run(backendScopeKey(null, profileKey), () => ensureBackend(profile, { spawnPriority })) const connectionId = resolvedConnectionId(readDesktopConnectionsRegistry(), connection) return connectionId ? { ...connection, connectionId } : connection @@ -14728,13 +14821,16 @@ ipcMain.handle('hermes:connection', async (_event, profile) => { // forces a genuinely-local child when the v1 global mode is remote (the // registry 'local' entry always means this machine). ipcMain.handle('hermes:connection:for', async (_event, payload) => { - const { connectionId, profile } = payload && typeof payload === 'object' ? (payload as any) : ({} as any) + const { connectionId, profile, priority } = + payload && typeof payload === 'object' ? (payload as any) : ({} as any) const registry = readDesktopConnectionsRegistry() const id = String(connectionId || '').trim() || registry.primary + const spawnPriority = spawnPriorityFrom(priority) // Same single-owner claim as 'hermes:connection', keyed by the composite // (connectionId, profile) scope (#90812): concurrent registry dials for one // scope share the first spawn instead of bootstrapping duplicate remotes. - const connection = await backendDialClaims.run(backendScopeKey(id, profile), () => ensureRegistryBackend(id, profile)) + promoteInFlightLocalSpawn(backendScopeKey(id, profile), spawnPriority) + const connection = await backendDialClaims.run(backendScopeKey(id, profile), () => ensureRegistryBackend(id, profile, '', { spawnPriority })) return { ...connection, connectionId: id, registryScoped: true } }) diff --git a/apps/desktop/electron/pool-spawn-coordinator.test.ts b/apps/desktop/electron/pool-spawn-coordinator.test.ts index 73b2291fe1..9d9a9d26ed 100644 --- a/apps/desktop/electron/pool-spawn-coordinator.test.ts +++ b/apps/desktop/electron/pool-spawn-coordinator.test.ts @@ -6,7 +6,7 @@ import { fileURLToPath } from 'node:url' import { test } from 'vitest' -import { LocalBackendSpawnCoordinator, releaseLocalBackendSlotAfterExit } from './pool-spawn-coordinator' +import { LocalBackendSlotWaitTimeoutError, LocalBackendSpawnCoordinator, releaseLocalBackendSlotAfterExit } from './pool-spawn-coordinator' const deferred = () => { let resolve!: () => void @@ -306,6 +306,159 @@ test('setLimit rejects a non-positive or fractional cap', () => { assert.equal(coordinator.limit, 2) }) + +test('cap 3: two background leases leave a reserved slot for foreground', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const bg1 = await coordinator.request('bg-1', { priority: 'background' }).acquired + const bg2 = await coordinator.request('bg-2', { priority: 'background' }).acquired + assert.equal(coordinator.activeCount, 2) + assert.equal(coordinator.queuedCount, 0) + + let fgGranted = false + const fgPromise = coordinator.request('fg', { priority: 'foreground' }).acquired.then(release => { + fgGranted = true + + return release + }) + + await flush() + assert.equal(fgGranted, true) + assert.equal(coordinator.activeCount, 3) + + const releaseFg = await fgPromise + bg1() + bg2() + releaseFg() + assert.equal(coordinator.activeCount, 0) +}) + +test('untagged acquire still fills the cap (foreground default)', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const releases = await Promise.all(['a', 'b', 'c'].map(key => coordinator.acquire(key))) + assert.equal(coordinator.activeCount, 3) + assert.equal(coordinator.queuedCount, 0) + for (const release of releases) { + release() + } + assert.equal(coordinator.activeCount, 0) +}) + +test('foreground is granted the reserved slot ahead of a background hydration queue', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const bgRunning = await Promise.all( + ['bg-run-1', 'bg-run-2'].map(key => coordinator.request(key, { priority: 'background' }).acquired) + ) + const queued = Array.from({ length: 20 }, (_, index) => + coordinator.request(`bg-wait-${index}`, { priority: 'background', timeoutMs: 5_000 }) + ) + await flush() + assert.equal(coordinator.activeCount, 2) + assert.equal(coordinator.queuedCount, 20) + + const started = Date.now() + const releaseFg = await coordinator.request('user-click', { priority: 'foreground', timeoutMs: 100 }).acquired + assert.ok(Date.now() - started < 80, 'foreground must not wait behind the background queue') + assert.equal(coordinator.activeCount, 3) + + for (const request of queued) { + request.cancel() + } + + releaseFg() + for (const release of bgRunning) { + release() + } + + await Promise.all(queued.map(request => request.acquired.then(() => undefined, () => undefined))) + assert.equal(coordinator.activeCount, 0) + assert.equal(coordinator.queuedCount, 0) +}) + +test('drain prefers a foreground waiter over an earlier background waiter', async () => { + const coordinator = new LocalBackendSpawnCoordinator(1) + const releaseHolder = await coordinator.acquire('holder') + const background = coordinator.request('background', { priority: 'background' }) + const foreground = coordinator.request('foreground', { priority: 'foreground' }) + await flush() + assert.equal(coordinator.queuedCount, 2) + + let backgroundEntered = false + let foregroundEntered = false + const backgroundGrant = background.acquired.then(release => { + backgroundEntered = true + + return release + }) + const foregroundGrant = foreground.acquired.then(release => { + foregroundEntered = true + + return release + }) + + releaseHolder() + await flush() + assert.equal(foregroundEntered, true) + assert.equal(backgroundEntered, false) + assert.equal(coordinator.activeCount, 1) + + const releaseForeground = await foregroundGrant + releaseForeground() + const releaseBackground = await backgroundGrant + assert.equal(backgroundEntered, true) + releaseBackground() + assert.equal(coordinator.activeCount, 0) +}) + +test('background slot-wait timeout is distinguishable; foreground keeps a user-facing message', async () => { + const coordinator = new LocalBackendSpawnCoordinator(1) + const releaseFirst = await coordinator.acquire('first') + + const background = coordinator.request('bg', { priority: 'background', timeoutMs: 10 }) + await assert.rejects(background.acquired, error => { + assert.ok(error instanceof LocalBackendSlotWaitTimeoutError) + assert.equal(error.name, 'LocalBackendSlotWaitTimeoutError') + assert.equal(error.priority, 'background') + assert.equal(error.silent, true) + assert.match(error.message, /timed out while waiting for a free slot/) + assert.match(error.message, /\(background\)/) + + return true + }) + + const foreground = coordinator.request('fg', { priority: 'foreground', timeoutMs: 10 }) + await assert.rejects(foreground.acquired, error => { + assert.ok(error instanceof Error) + assert.match(error.message, /timed out while waiting for a free slot/) + assert.doesNotMatch(error.message, /\(background\)/) + assert.notEqual(error.name, 'LocalBackendSlotWaitTimeoutError') + + return true + }) + + releaseFirst() + assert.equal(coordinator.activeCount, 0) +}) + +test('promoting a queued background waiter lets it take the reserved foreground slot', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const bg1 = await coordinator.request('bg-1', { priority: 'background' }).acquired + const bg2 = await coordinator.request('bg-2', { priority: 'background' }).acquired + const queued = coordinator.request('same-bot', { priority: 'background' }) + await flush() + assert.equal(coordinator.activeCount, 2) + assert.equal(coordinator.queuedCount, 1) + + assert.equal(queued.promote('foreground'), true) + const releasePromoted = await queued.acquired + assert.equal(coordinator.activeCount, 3) + assert.equal(coordinator.queuedCount, 0) + + releasePromoted() + bg1() + bg2() + assert.equal(coordinator.activeCount, 0) +}) + // ── main.ts wiring ────────────────────────────────────────────────────────── // The coordinator is only as good as the timeout main.ts hands it. A queued // ticket that outlives the renderer's backend-boot budget holds the pool key @@ -330,7 +483,10 @@ test('setLimit rejects a non-positive or fractional cap', () => { assert.ok(Number.isFinite(slotWait) && slotWait > 0, 'POOL_SLOT_WAIT_MS must be a literal in main.ts') assert.ok(Number.isFinite(bootBudget), 'BACKEND_BOOT_WAIT_TIMEOUT_MS must be a literal') assert.ok(slotWait < bootBudget, `slot wait ${slotWait}ms must be below the boot budget ${bootBudget}ms`) - assert.match(mainSource, /localBackendSpawnCoordinator\.request\(poolKey, \{ timeoutMs: POOL_SLOT_WAIT_MS \}\)/) + assert.match( + mainSource, + /localBackendSpawnCoordinator\.request\(poolKey, \{ timeoutMs: POOL_SLOT_WAIT_MS, priority: spawnPriority \}\)/ + ) assert.doesNotMatch(mainSource, /request\(poolKey, \{ timeoutMs: POOL_IDLE_MS \}\)/) }) diff --git a/apps/desktop/electron/pool-spawn-coordinator.ts b/apps/desktop/electron/pool-spawn-coordinator.ts index 8e565ed133..5738c32df7 100644 --- a/apps/desktop/electron/pool-spawn-coordinator.ts +++ b/apps/desktop/electron/pool-spawn-coordinator.ts @@ -1,17 +1,59 @@ export type ReleaseLocalBackendSlot = () => void +export type LocalBackendSpawnPriority = 'foreground' | 'background' + export type LocalBackendSpawnRequest = { acquired: Promise cancel: () => boolean + promote: (priority: LocalBackendSpawnPriority) => boolean } type Waiter = { key: string + priority: LocalBackendSpawnPriority resolve: (release: ReleaseLocalBackendSlot) => void reject: (error: Error) => void timer: ReturnType | null } +const SLOT_WAIT_TIMEOUT_MESSAGE = (key: string) => + `Local backend start for "${key}" timed out while waiting for a free slot.` + +/** + * Slot-wait timeout. Background hydrations set `silent` so call sites can fail + * quiet instead of toasting a user-visible backend-start failure. + */ +export class LocalBackendSlotWaitTimeoutError extends Error { + readonly priority: LocalBackendSpawnPriority + readonly silent: boolean + + constructor(key: string, priority: LocalBackendSpawnPriority) { + const suffix = priority === 'background' ? ' (background)' : '' + super(`${SLOT_WAIT_TIMEOUT_MESSAGE(key)}${suffix}`) + this.name = 'LocalBackendSlotWaitTimeoutError' + this.priority = priority + this.silent = priority === 'background' + } +} + +export function isBackgroundSlotWaitTimeout(error: unknown): boolean { + if (!error || typeof error !== 'object') { + return false + } + + const err = error as Error & { priority?: string; silent?: boolean } + + if (err.name === 'LocalBackendSlotWaitTimeoutError' && (err.silent === true || err.priority === 'background')) { + return true + } + + return ( + typeof err.message === 'string' && + err.message.includes('timed out while waiting for a free slot') && + (err.silent === true || err.priority === 'background' || err.message.includes('(background)')) + ) +} + export async function releaseLocalBackendSlotAfterExit( release: ReleaseLocalBackendSlot, waitForExit: () => Promise @@ -25,10 +67,15 @@ export async function releaseLocalBackendSlotAfterExit( * * A lease is acquired immediately before local start work and is held until * the child exits or the start fails. Remote descriptors never call request(). + * + * When the cap is at least 2, one slot is reserved for foreground (user-open) + * requests so background roster hydration cannot occupy the whole pool. + * Untagged acquire() is foreground, so existing cap tests still fill `limit`. */ export class LocalBackendSpawnCoordinator { #limit: number - #active = 0 + #activeForeground = 0 + #activeBackground = 0 #queue: Waiter[] = [] constructor(limit: number) { @@ -40,7 +87,7 @@ export class LocalBackendSpawnCoordinator { } get activeCount(): number { - return this.#active + return this.#activeForeground + this.#activeBackground } get limit(): number { @@ -66,39 +113,45 @@ export class LocalBackendSpawnCoordinator { return this.#queue.length } - request(key: string, options: { timeoutMs?: number } = {}): LocalBackendSpawnRequest { + request( + key: string, + options: { timeoutMs?: number; priority?: LocalBackendSpawnPriority } = {} + ): LocalBackendSpawnRequest { if (options.timeoutMs !== undefined && (!Number.isFinite(options.timeoutMs) || options.timeoutMs < 1)) { throw new RangeError('Local backend spawn timeout must be a positive number.') } - if (this.#active < this.#limit) { + const priority: LocalBackendSpawnPriority = options.priority === 'background' ? 'background' : 'foreground' + + if (this.#queue.length === 0 && this.#canGrant(priority)) { return { - acquired: Promise.resolve(this.#grant()), - cancel: () => false + acquired: Promise.resolve(this.#grant(priority)), + cancel: () => false, + promote: () => false } } let waiter!: Waiter const acquired = new Promise((resolve, reject) => { - waiter = { key, resolve, reject, timer: null } + waiter = { key, priority, resolve, reject, timer: null } this.#queue.push(waiter) if (options.timeoutMs !== undefined) { waiter.timer = setTimeout(() => { - this.#rejectWaiter( - waiter, - new Error(`Local backend start for "${key}" timed out while waiting for a free slot.`) - ) + this.#rejectWaiter(waiter, this.#timeoutError(waiter)) }, options.timeoutMs) waiter.timer.unref?.() } }) + this.#drain() + return { acquired, cancel: () => - this.#rejectWaiter(waiter, new Error(`Local backend start for "${key}" was cancelled while queued.`)) + this.#rejectWaiter(waiter, new Error(`Local backend start for "${key}" was cancelled while queued.`)), + promote: (nextPriority: LocalBackendSpawnPriority) => this.#promoteWaiter(waiter, nextPriority) } } @@ -106,6 +159,45 @@ export class LocalBackendSpawnCoordinator { return this.request(key).acquired } + #timeoutError(waiter: Waiter): Error { + if (waiter.priority === 'background') { + return new LocalBackendSlotWaitTimeoutError(waiter.key, 'background') + } + + return new Error(SLOT_WAIT_TIMEOUT_MESSAGE(waiter.key)) + } + + #backgroundLimit(): number { + return this.#limit >= 2 ? this.#limit - 1 : this.#limit + } + + #canGrant(priority: LocalBackendSpawnPriority): boolean { + if (this.activeCount >= this.#limit) { + return false + } + + if (priority === 'background' && this.#activeBackground >= this.#backgroundLimit()) { + return false + } + + return true + } + + #promoteWaiter(waiter: Waiter, priority: LocalBackendSpawnPriority): boolean { + if (!this.#queue.includes(waiter)) { + return false + } + + if (waiter.priority === priority) { + return false + } + + waiter.priority = priority + this.#drain() + + return true + } + #rejectWaiter(waiter: Waiter, error: Error): boolean { const index = this.#queue.indexOf(waiter) @@ -127,8 +219,13 @@ export class LocalBackendSpawnCoordinator { } } - #grant(): ReleaseLocalBackendSlot { - this.#active += 1 + #grant(priority: LocalBackendSpawnPriority): ReleaseLocalBackendSlot { + if (priority === 'background') { + this.#activeBackground += 1 + } else { + this.#activeForeground += 1 + } + let released = false return () => { @@ -137,17 +234,49 @@ export class LocalBackendSpawnCoordinator { } released = true - this.#active -= 1 + + if (priority === 'background') { + this.#activeBackground -= 1 + } else { + this.#activeForeground -= 1 + } + this.#drain() } } - /** Hand free slots to queued waiters while under the (possibly lowered) cap. */ + #takeWaiter(priority: LocalBackendSpawnPriority): Waiter | undefined { + const index = this.#queue.findIndex(waiter => waiter.priority === priority) + + if (index === -1) { + return undefined + } + + return this.#queue.splice(index, 1)[0] + } + + /** Hand free slots to queued waiters. Foreground waiters always go first. */ #drain(): void { - while (this.#active < this.#limit && this.#queue.length > 0) { - const next = this.#queue.shift()! + while (this.#canGrant('foreground')) { + const next = this.#takeWaiter('foreground') + + if (!next) { + break + } + this.#clearTimer(next) - next.resolve(this.#grant()) + next.resolve(this.#grant('foreground')) + } + + while (this.#canGrant('background')) { + const next = this.#takeWaiter('background') + + if (!next) { + break + } + + this.#clearTimer(next) + next.resolve(this.#grant('background')) } } } diff --git a/apps/desktop/electron/preload.ts b/apps/desktop/electron/preload.ts index fd9668752b..ca8bb45a28 100644 --- a/apps/desktop/electron/preload.ts +++ b/apps/desktop/electron/preload.ts @@ -18,7 +18,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', { // Launch-flag fact: the app was started with --local, so the renderer may // show the local-models surfaces. Static for the window's lifetime. localModelsEnabled: launchFlags?.localModels === true, - getConnection: profile => ipcRenderer.invoke('hermes:connection', profile), + getConnection: (profile, opts) => ipcRenderer.invoke('hermes:connection', profile, opts), // Registry-scoped backend resolution: { connectionId, profile } → descriptor. getConnectionFor: payload => ipcRenderer.invoke('hermes:connection:for', payload), getProfileRoutes: profiles => ipcRenderer.invoke('hermes:plugin-profile-routes', profiles), diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts index 9c0b509641..6a24a791c9 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/index.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/index.ts @@ -1001,9 +1001,11 @@ export function useSessionActions({ // dial the owning backend without moving $activeGatewayProfile. if ($showAllProfiles.get()) { if (resolvedConnectionId) { - await openGatewayForAgent(resolvedConnectionId, ownerRoute?.profile || sessionProfile || 'default') + await openGatewayForAgent(resolvedConnectionId, ownerRoute?.profile || sessionProfile || 'default', { + spawnPriority: 'foreground' + }) } else if (sessionProfile) { - await openGatewayForProfile(normalizeProfileKey(sessionProfile)) + await openGatewayForProfile(normalizeProfileKey(sessionProfile), { spawnPriority: 'foreground' }) } } else if (resolvedConnectionId) { await ensureGatewayAgent(resolvedConnectionId, ownerRoute?.profile || sessionProfile || 'default') diff --git a/apps/desktop/src/global.d.ts b/apps/desktop/src/global.d.ts index 0362b8ff12..43ff09810e 100644 --- a/apps/desktop/src/global.d.ts +++ b/apps/desktop/src/global.d.ts @@ -20,12 +20,16 @@ declare global { // Resolve a backend connection. Omit `profile` (or pass the primary) for // the window's backend; pass a named profile to lazily spawn/reuse that // profile's backend from the pool. - getConnection: (profile?: string | null) => Promise + getConnection: ( + profile?: string | null, + opts?: { priority?: 'foreground' | 'background' } + ) => Promise // Registry-scoped backend resolution: dial (connectionId, profile). An // empty/local connectionId delegates to the legacy getConnection path. getConnectionFor?: (payload: { connectionId?: null | string profile?: null | string + priority?: 'foreground' | 'background' }) => Promise // Registry-scoped fresh WS URL (same result contract as getGatewayWsUrl). getGatewayWsUrlFor?: (payload: { diff --git a/apps/desktop/src/sdk/index.ts b/apps/desktop/src/sdk/index.ts index bf28c8fb16..e69550dac7 100644 --- a/apps/desktop/src/sdk/index.ts +++ b/apps/desktop/src/sdk/index.ts @@ -911,11 +911,14 @@ export const host = { // not the registry-secondary path openGatewayForAgent takes for a 'local' // connection id. Behavior for a plain local open is unchanged. const dial = explicitRoute - ? () => openGatewayForAgent(explicitRoute.connectionId, explicitRoute.profile) + ? () => + openGatewayForAgent(explicitRoute.connectionId, explicitRoute.profile, { + spawnPriority: 'foreground' + }) : plan.switchWorkspace ? () => ensureGatewayProfile(plan.switchWorkspace as string) : plan.dialWithoutSwitching - ? () => openGatewayForProfile(plan.dialWithoutSwitching as string) + ? () => openGatewayForProfile(plan.dialWithoutSwitching as string, { spawnPriority: 'foreground' }) : null if (dial) { diff --git a/apps/desktop/src/sdk/profile-routing.test.ts b/apps/desktop/src/sdk/profile-routing.test.ts index 83328ad120..202b9fbaef 100644 --- a/apps/desktop/src/sdk/profile-routing.test.ts +++ b/apps/desktop/src/sdk/profile-routing.test.ts @@ -607,7 +607,7 @@ describe('profile-aware plugin session opens', () => { await host.openSession('remote-chat', { route }) - expect(openGatewayForAgent).toHaveBeenCalledWith('source-a', 'default') + expect(openGatewayForAgent).toHaveBeenCalledWith('source-a', 'default', expect.objectContaining({ spawnPriority: 'foreground' })) expect(ensureGatewayProfile).not.toHaveBeenCalled() expect(setShowAllProfiles).toHaveBeenCalledWith(true) expect($activeGatewayProfile.get()).toBe('remote-worker') @@ -1158,7 +1158,7 @@ describe('profile-aware plugin session opens', () => { }) expect(ensureGatewayProfile).not.toHaveBeenCalled() - expect(openGatewayForProfile).toHaveBeenCalledWith('worker') + expect(openGatewayForProfile).toHaveBeenCalledWith('worker', expect.objectContaining({ spawnPriority: 'foreground' })) expect(setShowAllProfiles).toHaveBeenCalledWith(true) expect($activeGatewayProfile.get()).toBe('default') }) @@ -1169,7 +1169,7 @@ describe('profile-aware plugin session opens', () => { await host.openSession('bot-chat', { profile: 'worker' }) expect(ensureGatewayProfile).not.toHaveBeenCalled() - expect(openGatewayForProfile).toHaveBeenCalledWith('worker') + expect(openGatewayForProfile).toHaveBeenCalledWith('worker', expect.objectContaining({ spawnPriority: 'foreground' })) expect(setShowAllProfiles).toHaveBeenCalledWith(true) expect($activeGatewayProfile.get()).toBe('default') }) diff --git a/apps/desktop/src/store/gateway.ts b/apps/desktop/src/store/gateway.ts index 5bf36abaea..bb0f819bce 100644 --- a/apps/desktop/src/store/gateway.ts +++ b/apps/desktop/src/store/gateway.ts @@ -20,6 +20,27 @@ import { setConnection, setGatewayState } from '@/store/session' const normKey = (profile: string | null | undefined): string => (profile ?? '').trim() || 'default' +type SpawnPriority = 'foreground' | 'background' + +function isBackgroundSlotWaitTimeout(error: unknown): boolean { + if (!(error instanceof Error)) { + return false + } + + const extra = error as Error & { priority?: string; silent?: boolean } + + return ( + extra.name === 'LocalBackendSlotWaitTimeoutError' || + extra.silent === true || + extra.priority === 'background' || + (error.message.includes('timed out while waiting for a free slot') && error.message.includes('(background)')) + ) +} + +function connectionPriorityOpts(priority: SpawnPriority): { priority: 'foreground' } | undefined { + return priority === 'foreground' ? { priority: 'foreground' } : undefined +} + // Read connection state through a call so TS control-flow analysis doesn't // narrow the getter to a constant across guards (it genuinely changes). const isOpen = (gateway: HermesGateway | null): boolean => gateway?.connectionState === 'open' @@ -489,7 +510,7 @@ function clearTimer(entry: Secondary): void { } } -async function openSecondary(entry: Secondary): Promise { +async function openSecondary(entry: Secondary, spawnPriority: SpawnPriority = 'background'): Promise { const desktop = window.hermesDesktop if (!desktop) { @@ -497,6 +518,18 @@ async function openSecondary(entry: Secondary): Promise { } if (entry.connectPromise) { + if (spawnPriority === 'foreground') { + // Hydration may already own this dial as a background slot wait. Kick a + // foreground IPC so main can promote it onto the reserved slot. + void (entry.connectionId && desktop.getConnectionFor + ? desktop.getConnectionFor({ + connectionId: entry.connectionId, + profile: entry.profile, + priority: 'foreground' + }) + : desktop.getConnection(entry.profile, { priority: 'foreground' })) + } + await entry.connectPromise return @@ -540,18 +573,33 @@ async function openSecondary(entry: Secondary): Promise { // this secondary (SSH terminal, messaging DELETE, session send, …) never // settles either. Bound the same way use-gateway-boot.ts bounds the // primary's equivalent awaits. - const conn = - entry.connectionId && desktop.getConnectionFor - ? await withTimeout( - desktop.getConnectionFor({ connectionId: entry.connectionId, profile: entry.profile }), - RECONNECT_ATTEMPT_TIMEOUT_MS, - `Timed out connecting to profile "${entry.profile}"` - ) - : await withTimeout( - desktop.getConnection(entry.profile), - RECONNECT_ATTEMPT_TIMEOUT_MS, - `Timed out connecting to profile "${entry.profile}"` - ) + const conn = await (async () => { + try { + return entry.connectionId && desktop.getConnectionFor + ? await withTimeout( + desktop.getConnectionFor({ + connectionId: entry.connectionId, + profile: entry.profile, + ...(connectionPriorityOpts(spawnPriority) ?? {}) + }), + RECONNECT_ATTEMPT_TIMEOUT_MS, + `Timed out connecting to profile "${entry.profile}"` + ) + : await withTimeout( + spawnPriority === 'foreground' + ? desktop.getConnection(entry.profile, { priority: 'foreground' }) + : desktop.getConnection(entry.profile), + RECONNECT_ATTEMPT_TIMEOUT_MS, + `Timed out connecting to profile "${entry.profile}"` + ) + } catch (error) { + if (spawnPriority !== 'foreground' && isBackgroundSlotWaitTimeout(error)) { + throw error + } + + throw error + } + })() entry.connection = conn @@ -781,7 +829,8 @@ async function sharedPrimaryRoute(profile: string): Promise { // request-scope flag; dedicated local/remote profiles use their pooled socket. async function gatewayForProfile( profile: string, - leaseRequest = false + leaseRequest = false, + spawnPriority: SpawnPriority = 'background' ): Promise<{ gateway: HermesGateway | null; key: string; release: () => void; scopeProfile: boolean }> { const key = normKey(profile) const noRelease = () => undefined @@ -840,7 +889,7 @@ async function gatewayForProfile( try { if (!isOpen(entry.gateway)) { - await openSecondary(entry) + await openSecondary(entry, spawnPriority) } } catch (error) { release() @@ -1300,8 +1349,11 @@ function releaseTerminalTurnLease(scope: string, event: GatewayEvent): void { // it. No scheduleReconnect on failure: a hover is speculative, so a dead // backend must not start a background retry loop — the real switch owns retry // and error UX. An already-open (or primary) profile is a no-op. -export async function openGatewayForProfile(profile: string): Promise { - await gatewayForProfile(profile) +export async function openGatewayForProfile( + profile: string, + { spawnPriority = 'background' }: { spawnPriority?: SpawnPriority } = {} +): Promise { + await gatewayForProfile(profile, false, spawnPriority) } // ── Connection-scoped agents (multi-source roster) ───────────────────────── @@ -1321,12 +1373,15 @@ export async function openGatewayForProfile(profile: string): Promise { export async function openGatewayForAgent( connectionId: null | string, profile: string, - { activationLease = false }: { activationLease?: boolean } = {} + { + activationLease = false, + spawnPriority = 'background' + }: { activationLease?: boolean; spawnPriority?: SpawnPriority } = {} ): Promise { const scope = registryBackendScopeKey(connectionId, profile) if (scope === normKey(profile) || isPrimaryRegistryRoute(connectionId, profile)) { - return openGatewayForProfile(profile) + return openGatewayForProfile(profile, { spawnPriority }) } if (await isAttachedSharedRemote(connectionId, profile)) { @@ -1356,7 +1411,7 @@ export async function openGatewayForAgent( } try { - await openSecondary(entry) + await openSecondary(entry, spawnPriority) } catch (error) { if (activationLease) { entry.activationLeaseUntil = 0 @@ -1412,7 +1467,7 @@ export async function ensureGatewayForAgent( entry.reconnectAttempt = 0 try { - await openSecondary(entry) + await openSecondary(entry, 'foreground') } catch { scheduleReconnect(entry) } @@ -1489,7 +1544,7 @@ export async function ensureGatewayForProfile(profile: string): Promise { entry.reconnectAttempt = 0 try { - await openSecondary(entry) + await openSecondary(entry, 'foreground') } catch (error) { // #81094: a failed secondary dial must NOT fall through to setActive() // with a closed socket — that silently routes the user's messages to the diff --git a/apps/desktop/src/store/profile.ts b/apps/desktop/src/store/profile.ts index b608f165f0..44d4d6b1ca 100644 --- a/apps/desktop/src/store/profile.ts +++ b/apps/desktop/src/store/profile.ts @@ -603,7 +603,10 @@ export async function openGatewayAgent(connectionId: string, profile: string): P return } - await openGatewayForAgent(connection, normalizeProfileKey(profile), { activationLease: true }) + await openGatewayForAgent(connection, normalizeProfileKey(profile), { + activationLease: true, + spawnPriority: 'foreground' + }) } // Activate a connection-scoped agent's gateway — the (connectionId, profile) From d196983b5dc18298de49d05934617b68a4edbc8f Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Fri, 4 Sep 2026 01:32:24 +0000 Subject: [PATCH 068/276] fix(desktop): observe foreground promotion failures Co-authored-by: Josh Tsai --- apps/desktop/src/store/gateway.ts | 18 +++++++++++------- 1 file changed, 11 insertions(+), 7 deletions(-) diff --git a/apps/desktop/src/store/gateway.ts b/apps/desktop/src/store/gateway.ts index bb0f819bce..4d864a4a5d 100644 --- a/apps/desktop/src/store/gateway.ts +++ b/apps/desktop/src/store/gateway.ts @@ -521,13 +521,17 @@ async function openSecondary(entry: Secondary, spawnPriority: SpawnPriority = 'b if (spawnPriority === 'foreground') { // Hydration may already own this dial as a background slot wait. Kick a // foreground IPC so main can promote it onto the reserved slot. - void (entry.connectionId && desktop.getConnectionFor - ? desktop.getConnectionFor({ - connectionId: entry.connectionId, - profile: entry.profile, - priority: 'foreground' - }) - : desktop.getConnection(entry.profile, { priority: 'foreground' })) + void ( + ( + entry.connectionId && desktop.getConnectionFor + ? desktop.getConnectionFor({ + connectionId: entry.connectionId, + profile: entry.profile, + priority: 'foreground' + }) + : desktop.getConnection(entry.profile, { priority: 'foreground' }) + ).catch(() => undefined) + ) } await entry.connectPromise From 6230e9767d3c606d4112a2d1130cfff0babb7a24 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Fri, 4 Sep 2026 01:36:15 +0000 Subject: [PATCH 069/276] style(desktop): format promotion rejection handler Co-authored-by: Josh Tsai --- apps/desktop/src/store/gateway.ts | 18 ++++++++---------- 1 file changed, 8 insertions(+), 10 deletions(-) diff --git a/apps/desktop/src/store/gateway.ts b/apps/desktop/src/store/gateway.ts index 4d864a4a5d..18c3bc59ea 100644 --- a/apps/desktop/src/store/gateway.ts +++ b/apps/desktop/src/store/gateway.ts @@ -522,16 +522,14 @@ async function openSecondary(entry: Secondary, spawnPriority: SpawnPriority = 'b // Hydration may already own this dial as a background slot wait. Kick a // foreground IPC so main can promote it onto the reserved slot. void ( - ( - entry.connectionId && desktop.getConnectionFor - ? desktop.getConnectionFor({ - connectionId: entry.connectionId, - profile: entry.profile, - priority: 'foreground' - }) - : desktop.getConnection(entry.profile, { priority: 'foreground' }) - ).catch(() => undefined) - ) + entry.connectionId && desktop.getConnectionFor + ? desktop.getConnectionFor({ + connectionId: entry.connectionId, + profile: entry.profile, + priority: 'foreground' + }) + : desktop.getConnection(entry.profile, { priority: 'foreground' }) + ).catch(() => undefined) } await entry.connectPromise From 49ab5e2b732be529781329e89e9442bb20041aab Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 13:54:03 +0530 Subject: [PATCH 070/276] fix(desktop): consume foreground spawn marks and log only real slot waits Follow-ups to the #102496 salvage in the Electron main process: - pendingForegroundSpawns leaked: promoteInFlightLocalSpawn marked the key even when the pool entry already existed (the common click path), and nothing consumed it. After that backend was reaped, the next dial for the key - normally 10 s roster hydration - spawned as foreground and sat in the reserved slot. Mark only when no entry exists yet; spawnPoolBackend consumes the mark before any early return (remote route included) so it never outlives the dial. - The "waiting for a free local slot" log fired for a foreground request that was granted the reserved slot immediately (condition was queuedCount > 0 after request()). The request now reports `queued`; log only then. - One promotePoolEntry() and one logPoolSpawnFailure() replace three copies of the promote snippet and two copies of the background/foreground log branch; spawnPoolBackend reads entry.spawnPriority instead of a second opts channel; isBackgroundSlotWaitTimeout is an instanceof check (same process as the class, no duck typing). --- apps/desktop/electron/main.ts | 142 +++++++++--------- .../electron/pool-spawn-coordinator.test.ts | 50 +++++- .../electron/pool-spawn-coordinator.ts | 24 +-- 3 files changed, 124 insertions(+), 92 deletions(-) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 40df34ff80..2a0e8c0e82 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -1494,44 +1494,59 @@ const localBackendSpawnCoordinator = new LocalBackendSpawnCoordinator(poolLimits // the queued ticket fails before the renderer does and the user sees why. const POOL_SLOT_WAIT_MS = 30_000 -function spawnPriorityFrom(value): LocalBackendSpawnPriority { +function spawnPriorityFrom(value: unknown): LocalBackendSpawnPriority { return value === 'foreground' ? 'foreground' : 'background' } -const pendingForegroundSpawns = new Set() +// Foreground intent for a dial whose pool entry does not exist yet: a user +// click that joins an in-flight backendDialClaims claim never re-enters +// ensureBackend(), and the claim owner may still be awaiting poolStopper / +// registry resolution before backendPool.set(). spawnPoolBackend() consumes the +// mark on every path (local or remote) so it cannot outlive the dial. +const pendingForegroundSpawns = new Set() -function markForegroundSpawn(poolKey): void { - if (poolKey) { - pendingForegroundSpawns.add(String(poolKey)) +function takeForegroundSpawn(...poolKeys: string[]): boolean { + let marked = false + + for (const poolKey of poolKeys) { + marked = pendingForegroundSpawns.delete(poolKey) || marked + } + + return marked +} + +// Upgrade a pooled entry (running, spawning, or queued for a slot) to +// foreground so a queued slot wait can take the reserved foreground slot. +function promotePoolEntry(entry: any): void { + entry.spawnPriority = 'foreground' + entry.localBackendSpawnRequest?.promote?.('foreground') +} + +// Land a spawn failure in desktop.log. A background slot-wait timeout is +// routine under a saturated pool (the next hydration pass retries), so it is +// logged as such instead of as a backend-start failure. +function logPoolSpawnFailure(label: string, error: unknown): void { + if (isBackgroundSlotWaitTimeout(error)) { + rememberLog(`Profile backend ${label} slot wait timed out (background); will retry on the next hydration`) + } else { + rememberLog( + `Hermes backend for profile ${label} failed to start: ${error instanceof Error ? error.message : String(error)}` + ) } } -function takeForegroundSpawn(poolKey): boolean { - const key = String(poolKey || '') - - if (!key || !pendingForegroundSpawns.has(key)) { - return false - } - - pendingForegroundSpawns.delete(key) - - return true -} - -function promoteInFlightLocalSpawn(poolKey, spawnPriority: LocalBackendSpawnPriority): void { +function promoteInFlightLocalSpawn(poolKey: string, spawnPriority: LocalBackendSpawnPriority): void { if (spawnPriority !== 'foreground') { return } - markForegroundSpawn(poolKey) const existing = backendPool.get(poolKey) - if (!existing) { - return + if (existing) { + promotePoolEntry(existing) + } else { + pendingForegroundSpawns.add(poolKey) } - - existing.spawnPriority = 'foreground' - existing.localBackendSpawnRequest?.promote?.('foreground') } function poolMaxBackends() { @@ -11460,8 +11475,7 @@ async function ensureBackend(profile, opts: { spawnPriority?: LocalBackendSpawnP existing.lastActiveAt = Date.now() if (spawnPriority === 'foreground') { - existing.spawnPriority = 'foreground' - existing.localBackendSpawnRequest?.promote?.('foreground') + promotePoolEntry(existing) } const connection = await existing.connectionPromise @@ -11485,19 +11499,11 @@ async function ensureBackend(profile, opts: { spawnPriority?: LocalBackendSpawnP spawnPriority } - entry.connectionPromise = spawnPoolBackend(key, entry, { spawnPriority }).catch(async error => { + entry.connectionPromise = spawnPoolBackend(key, entry).catch(async error => { // Land the failure in desktop.log: without this a spawn that dies before // its child exists (guard rejection, runtime resolution) leaves no trace // beyond renderer-side rejections users never see in a bundle. - if (isBackgroundSlotWaitTimeout(error)) { - rememberLog( - `Profile backend "${key}" slot wait timed out (background); will retry on the next hydration` - ) - } else { - rememberLog( - `Hermes backend for profile "${key}" failed to start: ${error instanceof Error ? error.message : String(error)}` - ) - } + logPoolSpawnFailure(`"${key}"`, error) await teardownFailedLocalBackend(key, entry) throw error @@ -11647,8 +11653,7 @@ async function ensureRegistryBackend( existingLocal.lastActiveAt = Date.now() if (spawnPriority === 'foreground') { - existingLocal.spawnPriority = 'foreground' - existingLocal.localBackendSpawnRequest?.promote?.('foreground') + promotePoolEntry(existingLocal) } return existingLocal.connectionPromise @@ -11671,20 +11676,11 @@ async function ensureRegistryBackend( localEntry.connectionPromise = spawnPoolBackend(profileKey, localEntry, { forceLocal: true, - poolKey: localRoute.poolKey, - spawnPriority + poolKey: localRoute.poolKey }).catch(async error => { // Same trace rule as the v1 pool path: a forced-local child whose spawn // rejects before the child exists must still land in desktop.log. - if (isBackgroundSlotWaitTimeout(error)) { - rememberLog( - `Profile backend "${profileKey}" (forced-local) slot wait timed out (background); will retry on the next hydration` - ) - } else { - rememberLog( - `Hermes backend for profile "${profileKey}" (forced-local) failed to start: ${error instanceof Error ? error.message : String(error)}` - ) - } + logPoolSpawnFailure(`"${profileKey}" (forced-local)`, error) await teardownFailedLocalBackend(localRoute.poolKey, localEntry) throw error @@ -12487,13 +12483,15 @@ function teardownFailedLocalBackend(poolKey: string, entry: any): Promise // entry means THIS machine regardless of the v1 routing table); `opts.poolKey` // is the backendPool key when it differs from the profile name (composite // registry scopes) so the exit/error cleanup evicts the right entry. -async function spawnPoolBackend( - profile, - entry, - opts: { forceLocal?: boolean; poolKey?: string; spawnPriority?: LocalBackendSpawnPriority } = {} -) { +async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; poolKey?: string } = {}) { const poolKey = opts.poolKey || profile + // The caller stamped entry.spawnPriority from its own request; a foreground + // dial that joined the claim before this entry existed left a mark instead. + if (takeForegroundSpawn(poolKey, profile)) { + entry.spawnPriority = 'foreground' + } + await reapOrphanedBackendsOnce() profileDeletionGate.assertCanStart(profile) @@ -12526,21 +12524,17 @@ async function spawnPoolBackend( // pool-idle window (10 min) would hold the pool key hostage and every // later click on the profile would join that stale wait. Failing here // surfaces the "all N slots busy" reason instead of a generic boot timeout. - const markedKey = takeForegroundSpawn(poolKey) - const markedProfile = takeForegroundSpawn(profile) - const spawnPriority = - entry.spawnPriority === 'foreground' || - opts.spawnPriority === 'foreground' || - markedKey || - markedProfile - ? 'foreground' - : 'background' - entry.spawnPriority = spawnPriority - const spawnRequest = localBackendSpawnCoordinator.request(poolKey, { timeoutMs: POOL_SLOT_WAIT_MS, priority: spawnPriority }) + const spawnPriority: LocalBackendSpawnPriority = spawnPriorityFrom(entry.spawnPriority) + + const spawnRequest = localBackendSpawnCoordinator.request(poolKey, { + timeoutMs: POOL_SLOT_WAIT_MS, + priority: spawnPriority + }) + entry.localBackendSlotKey = poolKey entry.localBackendSpawnRequest = spawnRequest - if (localBackendSpawnCoordinator.queuedCount > 0) { + if (spawnRequest.queued) { rememberLog( `Profile backend "${profile}" waiting for a free local slot (${localBackendSpawnCoordinator.activeCount}/${poolMaxBackends()} busy, ${localBackendSpawnCoordinator.queuedCount} queued)` ) @@ -14806,11 +14800,15 @@ ipcMain.handle('hermes:connection', async (_event, profile, extra) => { // both land here. The claim key mirrors ensureBackend()'s own profile // normalization so every spelling of the primary coalesces onto one dial. const profileKey = profile && String(profile).trim() ? String(profile).trim() : primaryProfileKey() - const spawnPriority = spawnPriorityFrom(extra && typeof extra === 'object' ? extra.priority : undefined) + const spawnPriority = spawnPriorityFrom(extra?.priority) // A user click may join an in-flight hydration claim; promote the queued // slot wait before coalescing so it can take the reserved foreground slot. promoteInFlightLocalSpawn(profileKey, spawnPriority) - const connection = await backendDialClaims.run(backendScopeKey(null, profileKey), () => ensureBackend(profile, { spawnPriority })) + + const connection = await backendDialClaims.run(backendScopeKey(null, profileKey), () => + ensureBackend(profile, { spawnPriority }) + ) + const connectionId = resolvedConnectionId(readDesktopConnectionsRegistry(), connection) return connectionId ? { ...connection, connectionId } : connection @@ -14821,8 +14819,7 @@ ipcMain.handle('hermes:connection', async (_event, profile, extra) => { // forces a genuinely-local child when the v1 global mode is remote (the // registry 'local' entry always means this machine). ipcMain.handle('hermes:connection:for', async (_event, payload) => { - const { connectionId, profile, priority } = - payload && typeof payload === 'object' ? (payload as any) : ({} as any) + const { connectionId, profile, priority } = payload && typeof payload === 'object' ? (payload as any) : ({} as any) const registry = readDesktopConnectionsRegistry() const id = String(connectionId || '').trim() || registry.primary const spawnPriority = spawnPriorityFrom(priority) @@ -14830,7 +14827,10 @@ ipcMain.handle('hermes:connection:for', async (_event, payload) => { // (connectionId, profile) scope (#90812): concurrent registry dials for one // scope share the first spawn instead of bootstrapping duplicate remotes. promoteInFlightLocalSpawn(backendScopeKey(id, profile), spawnPriority) - const connection = await backendDialClaims.run(backendScopeKey(id, profile), () => ensureRegistryBackend(id, profile, '', { spawnPriority })) + + const connection = await backendDialClaims.run(backendScopeKey(id, profile), () => + ensureRegistryBackend(id, profile, '', { spawnPriority }) + ) return { ...connection, connectionId: id, registryScoped: true } }) diff --git a/apps/desktop/electron/pool-spawn-coordinator.test.ts b/apps/desktop/electron/pool-spawn-coordinator.test.ts index 9d9a9d26ed..b7cf4ee30e 100644 --- a/apps/desktop/electron/pool-spawn-coordinator.test.ts +++ b/apps/desktop/electron/pool-spawn-coordinator.test.ts @@ -6,7 +6,11 @@ import { fileURLToPath } from 'node:url' import { test } from 'vitest' -import { LocalBackendSlotWaitTimeoutError, LocalBackendSpawnCoordinator, releaseLocalBackendSlotAfterExit } from './pool-spawn-coordinator' +import { + LocalBackendSlotWaitTimeoutError, + LocalBackendSpawnCoordinator, + releaseLocalBackendSlotAfterExit +} from './pool-spawn-coordinator' const deferred = () => { let resolve!: () => void @@ -306,7 +310,6 @@ test('setLimit rejects a non-positive or fractional cap', () => { assert.equal(coordinator.limit, 2) }) - test('cap 3: two background leases leave a reserved slot for foreground', async () => { const coordinator = new LocalBackendSpawnCoordinator(3) const bg1 = await coordinator.request('bg-1', { priority: 'background' }).acquired @@ -315,6 +318,7 @@ test('cap 3: two background leases leave a reserved slot for foreground', async assert.equal(coordinator.queuedCount, 0) let fgGranted = false + const fgPromise = coordinator.request('fg', { priority: 'foreground' }).acquired.then(release => { fgGranted = true @@ -337,20 +341,25 @@ test('untagged acquire still fills the cap (foreground default)', async () => { const releases = await Promise.all(['a', 'b', 'c'].map(key => coordinator.acquire(key))) assert.equal(coordinator.activeCount, 3) assert.equal(coordinator.queuedCount, 0) + for (const release of releases) { release() } + assert.equal(coordinator.activeCount, 0) }) test('foreground is granted the reserved slot ahead of a background hydration queue', async () => { const coordinator = new LocalBackendSpawnCoordinator(3) + const bgRunning = await Promise.all( ['bg-run-1', 'bg-run-2'].map(key => coordinator.request(key, { priority: 'background' }).acquired) ) + const queued = Array.from({ length: 20 }, (_, index) => coordinator.request(`bg-wait-${index}`, { priority: 'background', timeoutMs: 5_000 }) ) + await flush() assert.equal(coordinator.activeCount, 2) assert.equal(coordinator.queuedCount, 20) @@ -365,11 +374,19 @@ test('foreground is granted the reserved slot ahead of a background hydration qu } releaseFg() + for (const release of bgRunning) { release() } - await Promise.all(queued.map(request => request.acquired.then(() => undefined, () => undefined))) + await Promise.all( + queued.map(request => + request.acquired.then( + () => undefined, + () => undefined + ) + ) + ) assert.equal(coordinator.activeCount, 0) assert.equal(coordinator.queuedCount, 0) }) @@ -384,11 +401,13 @@ test('drain prefers a foreground waiter over an earlier background waiter', asyn let backgroundEntered = false let foregroundEntered = false + const backgroundGrant = background.acquired.then(release => { backgroundEntered = true return release }) + const foregroundGrant = foreground.acquired.then(release => { foregroundEntered = true @@ -439,6 +458,29 @@ test('background slot-wait timeout is distinguishable; foreground keeps a user-f assert.equal(coordinator.activeCount, 0) }) +test('request() reports whether the caller actually waited behind the queue', async () => { + const coordinator = new LocalBackendSpawnCoordinator(3) + const bg1 = coordinator.request('bg-1', { priority: 'background' }) + const bg2 = coordinator.request('bg-2', { priority: 'background' }) + const bgWait = coordinator.request('bg-3', { priority: 'background' }) + assert.equal(bg1.queued, false) + assert.equal(bg2.queued, false) + assert.equal(bgWait.queued, true) + + // The reserved slot is free: a foreground request is granted immediately + // even though a background waiter is queued. + const fg = coordinator.request('fg', { priority: 'foreground' }) + assert.equal(fg.queued, false) + assert.equal(coordinator.activeCount, 3) + + bgWait.cancel() + await bgWait.acquired.catch(() => undefined) + ;(await fg.acquired)() + ;(await bg1.acquired)() + ;(await bg2.acquired)() + assert.equal(coordinator.activeCount, 0) +}) + test('promoting a queued background waiter lets it take the reserved foreground slot', async () => { const coordinator = new LocalBackendSpawnCoordinator(3) const bg1 = await coordinator.request('bg-1', { priority: 'background' }).acquired @@ -485,7 +527,7 @@ test('promoting a queued background waiter lets it take the reserved foreground assert.ok(slotWait < bootBudget, `slot wait ${slotWait}ms must be below the boot budget ${bootBudget}ms`) assert.match( mainSource, - /localBackendSpawnCoordinator\.request\(poolKey, \{ timeoutMs: POOL_SLOT_WAIT_MS, priority: spawnPriority \}\)/ + /localBackendSpawnCoordinator\.request\(poolKey, \{\s*timeoutMs: POOL_SLOT_WAIT_MS,\s*priority: spawnPriority\s*\}\)/ ) assert.doesNotMatch(mainSource, /request\(poolKey, \{ timeoutMs: POOL_IDLE_MS \}\)/) }) diff --git a/apps/desktop/electron/pool-spawn-coordinator.ts b/apps/desktop/electron/pool-spawn-coordinator.ts index 5738c32df7..5f80af332d 100644 --- a/apps/desktop/electron/pool-spawn-coordinator.ts +++ b/apps/desktop/electron/pool-spawn-coordinator.ts @@ -6,6 +6,8 @@ export type LocalBackendSpawnRequest = { acquired: Promise cancel: () => boolean promote: (priority: LocalBackendSpawnPriority) => boolean + /** False when the slot was granted without waiting behind the queue. */ + queued: boolean } type Waiter = { @@ -37,21 +39,7 @@ export class LocalBackendSlotWaitTimeoutError extends Error { } export function isBackgroundSlotWaitTimeout(error: unknown): boolean { - if (!error || typeof error !== 'object') { - return false - } - - const err = error as Error & { priority?: string; silent?: boolean } - - if (err.name === 'LocalBackendSlotWaitTimeoutError' && (err.silent === true || err.priority === 'background')) { - return true - } - - return ( - typeof err.message === 'string' && - err.message.includes('timed out while waiting for a free slot') && - (err.silent === true || err.priority === 'background' || err.message.includes('(background)')) - ) + return error instanceof LocalBackendSlotWaitTimeoutError && error.silent } export async function releaseLocalBackendSlotAfterExit( @@ -127,7 +115,8 @@ export class LocalBackendSpawnCoordinator { return { acquired: Promise.resolve(this.#grant(priority)), cancel: () => false, - promote: () => false + promote: () => false, + queued: false } } @@ -151,7 +140,8 @@ export class LocalBackendSpawnCoordinator { acquired, cancel: () => this.#rejectWaiter(waiter, new Error(`Local backend start for "${key}" was cancelled while queued.`)), - promote: (nextPriority: LocalBackendSpawnPriority) => this.#promoteWaiter(waiter, nextPriority) + promote: (nextPriority: LocalBackendSpawnPriority) => this.#promoteWaiter(waiter, nextPriority), + queued: this.#queue.includes(waiter) } } From 355058b80d99fd321d5e9c54d4f4dc3600ccc471 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 13:54:03 +0530 Subject: [PATCH 071/276] fix(desktop): tag the route probe of a user open as foreground too Every user open first probes its route (sharedPrimaryRoute / isAttachedSharedRemote) with getConnection / getConnectionFor, and only then dials the secondary. With #102496 only the second dial carried priority: 'foreground', so main started (or joined) the spawn as a background slot wait on the probe and the click still waited out the probe's 20 s RECONNECT_ATTEMPT_TIMEOUT_MS before promotion kicked in. Thread the priority into both probes; the activation doors (ensureGatewayForProfile / ensureGatewayForAgent) pass 'foreground' explicitly. Also drop the renderer-side isBackgroundSlotWaitTimeout + the try/catch whose two branches both rethrew: Electron rebuilds IPC rejections as a plain Error, so name/silent/priority never reached the renderer and the helper was dead. Test: gateway-spawn-priority.test.ts asserts every dial of a foreground open carries the tag and an untagged open never does (red on the #102496 head). --- .../src/store/gateway-spawn-priority.test.ts | 112 ++++++++++++++++++ apps/desktop/src/store/gateway.ts | 95 +++++++-------- 2 files changed, 157 insertions(+), 50 deletions(-) create mode 100644 apps/desktop/src/store/gateway-spawn-priority.test.ts diff --git a/apps/desktop/src/store/gateway-spawn-priority.test.ts b/apps/desktop/src/store/gateway-spawn-priority.test.ts new file mode 100644 index 0000000000..58ea0a5eb2 --- /dev/null +++ b/apps/desktop/src/store/gateway-spawn-priority.test.ts @@ -0,0 +1,112 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +// #102281: a user-initiated open must reach Electron main as a FOREGROUND dial +// on its FIRST IPC, not only on the secondary's connect. Every open first +// probes the route (sharedPrimaryRoute / isAttachedSharedRemote) with +// getConnection / getConnectionFor; if that probe is untagged, main starts the +// spawn as a background slot wait and the click waits out the probe's 20s +// timeout before anything promotes it. + +vi.mock('@/hermes', () => ({ + setApiRequestConnection: vi.fn(), + HermesGateway: class { + connectionState = 'closed' + connect = async (): Promise => { + this.connectionState = 'open' + } + close = (): void => { + this.connectionState = 'closed' + } + onEvent = vi.fn(() => () => {}) + onState = vi.fn(() => () => {}) + } +})) +vi.mock('@/store/session', () => ({ setConnection: vi.fn(), setGatewayState: vi.fn() })) +vi.mock('@/store/notify-baseline', () => ({ markNativeNotifyBaseline: vi.fn() })) + +const { + closeSecondaryGateways, + configureGatewayRegistry, + ensureGatewayForAgent, + ensureGatewayForProfile, + openGatewayForAgent, + openGatewayForProfile, + setPrimaryGateway +} = await import('./gateway') + +const conn = { + authMode: 'token', + baseUrl: 'https://homelab.invalid', + mode: 'remote', + profile: 'research', + token: 'fake-test-token', + wsUrl: 'wss://homelab.invalid/api/ws?token=fake-test-token' +} + +function installDesktop(): { getConnection: ReturnType; getConnectionFor: ReturnType } { + const stub = { + getConnection: vi.fn(async () => conn), + getConnectionFor: vi.fn(async () => conn) + } + + ;(window as unknown as { hermesDesktop: unknown }).hermesDesktop = stub + + return stub +} + +function priorities(mock: ReturnType, pick: (args: unknown[]) => unknown): unknown[] { + return mock.mock.calls.map(args => pick(args)) +} + +beforeEach(() => { + configureGatewayRegistry({ onEvent: vi.fn() }) + setPrimaryGateway({ connectionState: 'open' } as never, 'default') +}) + +afterEach(() => { + closeSecondaryGateways() + vi.clearAllMocks() + delete (window as unknown as { hermesDesktop?: unknown }).hermesDesktop +}) + +describe('user opens dial main as foreground from the first IPC (#102281)', () => { + it('ensureGatewayForProfile tags the route probe AND the connect dial', async () => { + const desktop = installDesktop() + + await ensureGatewayForProfile('research') + + const seen = priorities(desktop.getConnection, args => (args[1] as { priority?: string } | undefined)?.priority) + expect(seen.length).toBeGreaterThanOrEqual(2) + expect(seen.every(priority => priority === 'foreground')).toBe(true) + }) + + it('openGatewayForProfile without a priority never tags a dial as foreground', async () => { + const desktop = installDesktop() + + await openGatewayForProfile('research') + + const seen = priorities(desktop.getConnection, args => (args[1] as { priority?: string } | undefined)?.priority) + expect(seen.length).toBeGreaterThanOrEqual(1) + expect(seen.every(priority => priority === undefined)).toBe(true) + }) + + it('openGatewayForAgent forwards spawnPriority to every registry dial', async () => { + const desktop = installDesktop() + + await openGatewayForAgent('homelab', 'research', { spawnPriority: 'foreground' }) + + const seen = priorities(desktop.getConnectionFor, args => (args[0] as { priority?: string }).priority) + expect(seen.length).toBeGreaterThanOrEqual(1) + expect(seen.every(priority => priority === 'foreground')).toBe(true) + }) + + it('ensureGatewayForAgent is always a foreground open', async () => { + const desktop = installDesktop() + + await ensureGatewayForAgent('homelab', 'research') + + const seen = priorities(desktop.getConnectionFor, args => (args[0] as { priority?: string }).priority) + expect(seen.length).toBeGreaterThanOrEqual(1) + expect(seen.every(priority => priority === 'foreground')).toBe(true) + }) +}) diff --git a/apps/desktop/src/store/gateway.ts b/apps/desktop/src/store/gateway.ts index 18c3bc59ea..0812aec00f 100644 --- a/apps/desktop/src/store/gateway.ts +++ b/apps/desktop/src/store/gateway.ts @@ -20,25 +20,24 @@ import { setConnection, setGatewayState } from '@/store/session' const normKey = (profile: string | null | undefined): string => (profile ?? '').trim() || 'default' +// Spawn-slot priority handed to Electron main with every backend dial. A +// user-initiated open is 'foreground' and may take the pool's reserved slot; +// roster hydration, hover prewarm and untagged dials are 'background' — main's +// default, so background dials keep the pre-priority IPC payload shape. type SpawnPriority = 'foreground' | 'background' -function isBackgroundSlotWaitTimeout(error: unknown): boolean { - if (!(error instanceof Error)) { - return false - } - - const extra = error as Error & { priority?: string; silent?: boolean } - - return ( - extra.name === 'LocalBackendSlotWaitTimeoutError' || - extra.silent === true || - extra.priority === 'background' || - (error.message.includes('timed out while waiting for a free slot') && error.message.includes('(background)')) - ) +function dialPriority(spawnPriority: SpawnPriority): { priority: 'foreground' } | Record { + return spawnPriority === 'foreground' ? { priority: 'foreground' } : {} } -function connectionPriorityOpts(priority: SpawnPriority): { priority: 'foreground' } | undefined { - return priority === 'foreground' ? { priority: 'foreground' } : undefined +function dialProfile( + desktop: NonNullable, + profile: string, + spawnPriority: SpawnPriority +): Promise { + return spawnPriority === 'foreground' + ? desktop.getConnection(profile, { priority: 'foreground' }) + : desktop.getConnection(profile) } // Read connection state through a call so TS control-flow analysis doesn't @@ -323,7 +322,11 @@ function isPrimaryRegistryRoute(connectionId: null | string, profile: string): b * dials a second WebSocket at the same Tailscale URL, which accept/closes in * ~30ms (`messages=1`) and never runs `session.create` (#96493). Isolated * SSH/pooled backends (`sharedRemote: false`) still get their own secondary. */ -async function isAttachedSharedRemote(connectionId: null | string, profile: string): Promise { +async function isAttachedSharedRemote( + connectionId: null | string, + profile: string, + spawnPriority: SpawnPriority = 'background' +): Promise { const id = String(connectionId ?? '').trim() const key = normKey(profile) @@ -343,7 +346,7 @@ async function isAttachedSharedRemote(connectionId: null | string, profile: stri try { const conn = await withTimeout( - desktop.getConnectionFor({ connectionId: id, profile: key }), + desktop.getConnectionFor({ connectionId: id, profile: key, ...dialPriority(spawnPriority) }), RECONNECT_ATTEMPT_TIMEOUT_MS, `Timed out resolving shared-remote route for "${key}"` ) @@ -575,33 +578,22 @@ async function openSecondary(entry: Secondary, spawnPriority: SpawnPriority = 'b // this secondary (SSH terminal, messaging DELETE, session send, …) never // settles either. Bound the same way use-gateway-boot.ts bounds the // primary's equivalent awaits. - const conn = await (async () => { - try { - return entry.connectionId && desktop.getConnectionFor - ? await withTimeout( - desktop.getConnectionFor({ - connectionId: entry.connectionId, - profile: entry.profile, - ...(connectionPriorityOpts(spawnPriority) ?? {}) - }), - RECONNECT_ATTEMPT_TIMEOUT_MS, - `Timed out connecting to profile "${entry.profile}"` - ) - : await withTimeout( - spawnPriority === 'foreground' - ? desktop.getConnection(entry.profile, { priority: 'foreground' }) - : desktop.getConnection(entry.profile), - RECONNECT_ATTEMPT_TIMEOUT_MS, - `Timed out connecting to profile "${entry.profile}"` - ) - } catch (error) { - if (spawnPriority !== 'foreground' && isBackgroundSlotWaitTimeout(error)) { - throw error - } - - throw error - } - })() + const conn = + entry.connectionId && desktop.getConnectionFor + ? await withTimeout( + desktop.getConnectionFor({ + connectionId: entry.connectionId, + profile: entry.profile, + ...dialPriority(spawnPriority) + }), + RECONNECT_ATTEMPT_TIMEOUT_MS, + `Timed out connecting to profile "${entry.profile}"` + ) + : await withTimeout( + dialProfile(desktop, entry.profile, spawnPriority), + RECONNECT_ATTEMPT_TIMEOUT_MS, + `Timed out connecting to profile "${entry.profile}"` + ) entry.connection = conn @@ -802,7 +794,7 @@ function createSecondary(profile: string, connectionId: null | string = null): S // the second dial fails (tunnel/token are per-backend) and the closed socket // poisons the active gateway with "not connected" even though the primary is // open right next to it. -async function sharedPrimaryRoute(profile: string): Promise { +async function sharedPrimaryRoute(profile: string, spawnPriority: SpawnPriority = 'background'): Promise { const desktop = window.hermesDesktop if (!desktop) { @@ -814,8 +806,11 @@ async function sharedPrimaryRoute(profile: string): Promise { // like any other failure, not hang the route decision forever, since // every caller (gatewayForProfile → requestGatewayForProfile/Agent) awaits // this before it can fall back to dialing a secondary. + // This is the FIRST dial main sees for a user open, so it must already + // carry the foreground priority — otherwise the spawn it starts queues as + // background and the click waits out this probe before being promoted. const conn = await withTimeout( - desktop.getConnection(profile), + dialProfile(desktop, profile, spawnPriority), RECONNECT_ATTEMPT_TIMEOUT_MS, `Timed out resolving the shared-primary route for profile "${profile}"` ) @@ -841,7 +836,7 @@ async function gatewayForProfile( return { gateway: g.primaryGateway, key, release: noRelease, scopeProfile: false } } - if (await sharedPrimaryRoute(key)) { + if (await sharedPrimaryRoute(key, spawnPriority)) { return { gateway: g.primaryGateway, key, release: noRelease, scopeProfile: true } } @@ -1386,7 +1381,7 @@ export async function openGatewayForAgent( return openGatewayForProfile(profile, { spawnPriority }) } - if (await isAttachedSharedRemote(connectionId, profile)) { + if (await isAttachedSharedRemote(connectionId, profile, spawnPriority)) { if (!isOpen(g.primaryGateway)) { throw new Error('Hermes gateway unavailable') } @@ -1440,7 +1435,7 @@ export async function ensureGatewayForAgent( return !signal?.aborted } - if (await isAttachedSharedRemote(connectionId, profile)) { + if (await isAttachedSharedRemote(connectionId, profile, 'foreground')) { return Boolean(isOpen(g.primaryGateway) && !signal?.aborted) } @@ -1522,7 +1517,7 @@ export async function ensureGatewayForProfile(profile: string): Promise { // primary instead of dialing a doomed duplicate socket at the same // descriptor — $activeGatewayProfile still moves to `key`, so request // scoping and profile-aware surfaces behave identically. - if (await sharedPrimaryRoute(key)) { + if (await sharedPrimaryRoute(key, 'foreground')) { applyActive(g.primaryProfile, activationEpoch) return From 08f170aeab99d9206891cf8fa787c056b8b83f58 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:17:22 +0530 Subject: [PATCH 072/276] fix(desktop): clear an unconsumed foreground spawn mark when the dial settles spawnPoolBackend() is not on every dial path: a primary route (startHermes), a registry remote scope (connectRegistryBackend), a reused primary SSH backend, or a guard rejection all settle the claim without requesting a slot, so a foreground mark set for that dial stayed in pendingForegroundSpawns and would have upgraded the next background hydration spawn of the same key. applySpawnPriority() now returns the cleanup; both IPC handlers run it in a finally around the claim. The mark is also taken right before the slot request instead of at function entry, so the remote branch never consumes it. --- .../electron/backend-dial-claim.test.ts | 3 +- apps/desktop/electron/main.ts | 65 ++++++++++++------- 2 files changed, 45 insertions(+), 23 deletions(-) diff --git a/apps/desktop/electron/backend-dial-claim.test.ts b/apps/desktop/electron/backend-dial-claim.test.ts index cb3d6099ec..30c7e8195c 100644 --- a/apps/desktop/electron/backend-dial-claim.test.ts +++ b/apps/desktop/electron/backend-dial-claim.test.ts @@ -137,7 +137,8 @@ describe('main.ts wiring for #90812', () => { expect(handlerStart).toBeGreaterThan(-1) const body = mainSource.slice(handlerStart, handlerStart + 1_200) - expect(body).toContain('backendDialClaims.run(backendScopeKey(id, profile)') + expect(body).toContain('const scopeKey = backendScopeKey(id, profile)') + expect(body).toContain('backendDialClaims.run(scopeKey, ') expect(body).toContain("ensureRegistryBackend(id, profile, '', { spawnPriority })") }) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 2a0e8c0e82..ecbc3ebb71 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -1501,8 +1501,11 @@ function spawnPriorityFrom(value: unknown): LocalBackendSpawnPriority { // Foreground intent for a dial whose pool entry does not exist yet: a user // click that joins an in-flight backendDialClaims claim never re-enters // ensureBackend(), and the claim owner may still be awaiting poolStopper / -// registry resolution before backendPool.set(). spawnPoolBackend() consumes the -// mark on every path (local or remote) so it cannot outlive the dial. +// registry resolution before backendPool.set(). The local spawn takes the mark +// right before its slot request; the IPC handler that set it clears it once +// the claim settles, so a dial that never reaches a slot request (primary +// route, remote scope, a guard rejection) cannot leave it for a later +// hydration spawn of the same key to pick up. const pendingForegroundSpawns = new Set() function takeForegroundSpawn(...poolKeys: string[]): boolean { @@ -1535,18 +1538,24 @@ function logPoolSpawnFailure(label: string, error: unknown): void { } } -function promoteInFlightLocalSpawn(poolKey: string, spawnPriority: LocalBackendSpawnPriority): void { +// Apply foreground intent to the dial claim for `scopeKey`: an entry already +// in the pool is promoted directly, otherwise the intent is marked for the +// spawn the claim owner is about to start. Returns the cleanup that clears a +// mark the dial never consumed. +function applySpawnPriority(scopeKey: string, spawnPriority: LocalBackendSpawnPriority): () => void { if (spawnPriority !== 'foreground') { - return + return () => undefined } - const existing = backendPool.get(poolKey) + const existing = backendPool.get(scopeKey) if (existing) { promotePoolEntry(existing) } else { - pendingForegroundSpawns.add(poolKey) + pendingForegroundSpawns.add(scopeKey) } + + return () => void pendingForegroundSpawns.delete(scopeKey) } function poolMaxBackends() { @@ -12486,12 +12495,6 @@ function teardownFailedLocalBackend(poolKey: string, entry: any): Promise async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; poolKey?: string } = {}) { const poolKey = opts.poolKey || profile - // The caller stamped entry.spawnPriority from its own request; a foreground - // dial that joined the claim before this entry existed left a mark instead. - if (takeForegroundSpawn(poolKey, profile)) { - entry.spawnPriority = 'foreground' - } - await reapOrphanedBackendsOnce() profileDeletionGate.assertCanStart(profile) @@ -12524,6 +12527,12 @@ async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; po // pool-idle window (10 min) would hold the pool key hostage and every // later click on the profile would join that stale wait. Failing here // surfaces the "all N slots busy" reason instead of a generic boot timeout. + // The caller stamped entry.spawnPriority from its own request; a foreground + // dial that joined the claim before this entry existed left a mark instead. + if (takeForegroundSpawn(poolKey, profile)) { + entry.spawnPriority = 'foreground' + } + const spawnPriority: LocalBackendSpawnPriority = spawnPriorityFrom(entry.spawnPriority) const spawnRequest = localBackendSpawnCoordinator.request(poolKey, { @@ -14800,14 +14809,20 @@ ipcMain.handle('hermes:connection', async (_event, profile, extra) => { // both land here. The claim key mirrors ensureBackend()'s own profile // normalization so every spelling of the primary coalesces onto one dial. const profileKey = profile && String(profile).trim() ? String(profile).trim() : primaryProfileKey() + // A user click may join an in-flight hydration claim; the foreground intent + // is applied to that claim so its slot wait can take the reserved slot. const spawnPriority = spawnPriorityFrom(extra?.priority) - // A user click may join an in-flight hydration claim; promote the queued - // slot wait before coalescing so it can take the reserved foreground slot. - promoteInFlightLocalSpawn(profileKey, spawnPriority) - const connection = await backendDialClaims.run(backendScopeKey(null, profileKey), () => - ensureBackend(profile, { spawnPriority }) - ) + const scopeKey = backendScopeKey(null, profileKey) + const clearSpawnPriority = applySpawnPriority(scopeKey, spawnPriority) + + let connection + + try { + connection = await backendDialClaims.run(scopeKey, () => ensureBackend(profile, { spawnPriority })) + } finally { + clearSpawnPriority() + } const connectionId = resolvedConnectionId(readDesktopConnectionsRegistry(), connection) @@ -14823,14 +14838,20 @@ ipcMain.handle('hermes:connection:for', async (_event, payload) => { const registry = readDesktopConnectionsRegistry() const id = String(connectionId || '').trim() || registry.primary const spawnPriority = spawnPriorityFrom(priority) + // Same single-owner claim as 'hermes:connection', keyed by the composite // (connectionId, profile) scope (#90812): concurrent registry dials for one // scope share the first spawn instead of bootstrapping duplicate remotes. - promoteInFlightLocalSpawn(backendScopeKey(id, profile), spawnPriority) + const scopeKey = backendScopeKey(id, profile) + const clearSpawnPriority = applySpawnPriority(scopeKey, spawnPriority) - const connection = await backendDialClaims.run(backendScopeKey(id, profile), () => - ensureRegistryBackend(id, profile, '', { spawnPriority }) - ) + let connection + + try { + connection = await backendDialClaims.run(scopeKey, () => ensureRegistryBackend(id, profile, '', { spawnPriority })) + } finally { + clearSpawnPriority() + } return { ...connection, connectionId: id, registryScoped: true } }) From 6f4a822d9fa15735853a6292d48ac6ab371765c3 Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Sun, 6 Sep 2026 09:13:28 +0000 Subject: [PATCH 073/276] fmt(js): `npm run fix` on merge (#104156) Co-authored-by: github-actions[bot] --- apps/desktop/src/sdk/profile-routing.test.ts | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/apps/desktop/src/sdk/profile-routing.test.ts b/apps/desktop/src/sdk/profile-routing.test.ts index 202b9fbaef..c79f6081ed 100644 --- a/apps/desktop/src/sdk/profile-routing.test.ts +++ b/apps/desktop/src/sdk/profile-routing.test.ts @@ -607,7 +607,11 @@ describe('profile-aware plugin session opens', () => { await host.openSession('remote-chat', { route }) - expect(openGatewayForAgent).toHaveBeenCalledWith('source-a', 'default', expect.objectContaining({ spawnPriority: 'foreground' })) + expect(openGatewayForAgent).toHaveBeenCalledWith( + 'source-a', + 'default', + expect.objectContaining({ spawnPriority: 'foreground' }) + ) expect(ensureGatewayProfile).not.toHaveBeenCalled() expect(setShowAllProfiles).toHaveBeenCalledWith(true) expect($activeGatewayProfile.get()).toBe('remote-worker') @@ -1158,7 +1162,10 @@ describe('profile-aware plugin session opens', () => { }) expect(ensureGatewayProfile).not.toHaveBeenCalled() - expect(openGatewayForProfile).toHaveBeenCalledWith('worker', expect.objectContaining({ spawnPriority: 'foreground' })) + expect(openGatewayForProfile).toHaveBeenCalledWith( + 'worker', + expect.objectContaining({ spawnPriority: 'foreground' }) + ) expect(setShowAllProfiles).toHaveBeenCalledWith(true) expect($activeGatewayProfile.get()).toBe('default') }) @@ -1169,7 +1176,10 @@ describe('profile-aware plugin session opens', () => { await host.openSession('bot-chat', { profile: 'worker' }) expect(ensureGatewayProfile).not.toHaveBeenCalled() - expect(openGatewayForProfile).toHaveBeenCalledWith('worker', expect.objectContaining({ spawnPriority: 'foreground' })) + expect(openGatewayForProfile).toHaveBeenCalledWith( + 'worker', + expect.objectContaining({ spawnPriority: 'foreground' }) + ) expect(setShowAllProfiles).toHaveBeenCalledWith(true) expect($activeGatewayProfile.get()).toBe('default') }) From cca26097c5bdc538ee689ac23922a218e10f9e3a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 01:58:05 -0700 Subject: [PATCH 074/276] fix(desktop): status bar shows for every install, including ones that hid it under the old key MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The whole-bar preference lived at `hermes.desktop.statusbarVisible`. For a stretch (d399c164 → 120e465c) the atom's fallback was `false`, so any install that launched in that window persisted a hidden bar the user never chose, and flipping the fallback back to `true` only helped fresh stores. Move the preference to `hermes.desktop.statusbarVisible.v2` and do not seed it from v1: every existing install comes back to "on" once on update, and a hide made afterwards persists under the new key. Fresh installs are on by default as before. --- .../desktop/src/store/statusbar-prefs.test.ts | 32 +++++++++++++++++++ apps/desktop/src/store/statusbar-prefs.ts | 6 +++- 2 files changed, 37 insertions(+), 1 deletion(-) create mode 100644 apps/desktop/src/store/statusbar-prefs.test.ts diff --git a/apps/desktop/src/store/statusbar-prefs.test.ts b/apps/desktop/src/store/statusbar-prefs.test.ts new file mode 100644 index 0000000000..2d41618cd3 --- /dev/null +++ b/apps/desktop/src/store/statusbar-prefs.test.ts @@ -0,0 +1,32 @@ +import { beforeEach, describe, expect, it, vi } from 'vitest' + +const LEGACY_VISIBLE_KEY = 'hermes.desktop.statusbarVisible' + +const loadStore = () => import('./statusbar-prefs') + +describe('statusbar whole-bar visibility', () => { + beforeEach(() => { + window.localStorage.clear() + vi.resetModules() + }) + + it('shows the bar on a fresh install and for installs that hid it under the v1 key', async () => { + window.localStorage.setItem(LEGACY_VISIBLE_KEY, 'false') + + const { $statusbarVisible } = await loadStore() + + expect($statusbarVisible.get()).toBe(true) + }) + + it('still honours a hide made after the update', async () => { + const first = await loadStore() + + first.toggleStatusbarVisible() + expect(first.$statusbarVisible.get()).toBe(false) + + vi.resetModules() + const reloaded = await loadStore() + + expect(reloaded.$statusbarVisible.get()).toBe(false) + }) +}) diff --git a/apps/desktop/src/store/statusbar-prefs.ts b/apps/desktop/src/store/statusbar-prefs.ts index a82e91617d..16fc197353 100644 --- a/apps/desktop/src/store/statusbar-prefs.ts +++ b/apps/desktop/src/store/statusbar-prefs.ts @@ -1,7 +1,11 @@ import { Codecs, persistentAtom } from '@/lib/persisted' const STATUSBAR_HIDDEN_STORAGE_KEY = 'hermes.desktop.statusbarHidden' -const STATUSBAR_VISIBLE_STORAGE_KEY = 'hermes.desktop.statusbarVisible' +// v1 (`hermes.desktop.statusbarVisible`) shipped a stretch where the bar was +// opt-in, so many stores hold a `false` the user never chose. v2 is read fresh +// and the v1 key is deliberately NOT seeded from: every existing install comes +// back to "on" once, and hiding it again persists here. +const STATUSBAR_VISIBLE_STORAGE_KEY = 'hermes.desktop.statusbarVisible.v2' // Whole-bar visibility, VS Code's `workbench.statusBar.visible`. On by default. // Hiding it unmounts the bar (its 15s status poll goes with it), so the way back From 2bb9c4e6934cc871f207a1d59ff29666225581a6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:11:59 -0700 Subject: [PATCH 075/276] feat(desktop): approval-mode zap shows on the status bar by default Whether dangerous commands run unasked is state worth seeing at a glance, so the approval pill (yolo lightning) leaves STATUSBAR_HIDDEN_BY_DEFAULT. Existing stores were seeded with it hidden, so the hidden-set key moves to `hermes.desktop.statusbarHidden.v2`, seeded from v1 minus `approval-mode`: other customizations survive, the zap appears once on update, and hiding it again persists under the new key. --- .../app/shell/statusbar-visibility.test.tsx | 3 +- .../desktop/src/store/statusbar-prefs.test.ts | 26 +++++++++++++ apps/desktop/src/store/statusbar-prefs.ts | 38 +++++++++++++++---- 3 files changed, 58 insertions(+), 9 deletions(-) diff --git a/apps/desktop/src/app/shell/statusbar-visibility.test.tsx b/apps/desktop/src/app/shell/statusbar-visibility.test.tsx index 1a078ba38b..2010ed1cdf 100644 --- a/apps/desktop/src/app/shell/statusbar-visibility.test.tsx +++ b/apps/desktop/src/app/shell/statusbar-visibility.test.tsx @@ -57,11 +57,12 @@ describe('statusbar item visibility', () => { item('gateway-health', 'Gateway') ]) - for (const label of ['Cron', 'Webhooks', 'Agents', 'Terminal', 'Approvals']) { + for (const label of ['Cron', 'Webhooks', 'Agents', 'Terminal']) { expect(screen.queryByText(label)).toBeNull() } expect(screen.getByText('Gateway')).toBeTruthy() + expect(screen.getByText('Approvals')).toBeTruthy() }) it('shows an item once the user enables it from the bar context menu', async () => { diff --git a/apps/desktop/src/store/statusbar-prefs.test.ts b/apps/desktop/src/store/statusbar-prefs.test.ts index 2d41618cd3..a8f67417d9 100644 --- a/apps/desktop/src/store/statusbar-prefs.test.ts +++ b/apps/desktop/src/store/statusbar-prefs.test.ts @@ -30,3 +30,29 @@ describe('statusbar whole-bar visibility', () => { expect(reloaded.$statusbarVisible.get()).toBe(false) }) }) + +describe('statusbar hidden items', () => { + beforeEach(() => { + window.localStorage.clear() + vi.resetModules() + }) + + it('surfaces the approval pill for installs that hid it under the v1 defaults, keeping their other choices', async () => { + window.localStorage.setItem('hermes.desktop.statusbarHidden', JSON.stringify(['approval-mode', 'cron', 'gateway-health'])) + + const { $statusbarHiddenIds } = await loadStore() + + expect($statusbarHiddenIds.get()).toEqual(['cron', 'gateway-health']) + }) + + it('still honours hiding the approval pill after the update', async () => { + const first = await loadStore() + + first.setStatusbarItemVisible('approval-mode', false) + + vi.resetModules() + const reloaded = await loadStore() + + expect(reloaded.$statusbarHiddenIds.get()).toContain('approval-mode') + }) +}) diff --git a/apps/desktop/src/store/statusbar-prefs.ts b/apps/desktop/src/store/statusbar-prefs.ts index 16fc197353..c1857368c4 100644 --- a/apps/desktop/src/store/statusbar-prefs.ts +++ b/apps/desktop/src/store/statusbar-prefs.ts @@ -1,6 +1,12 @@ import { Codecs, persistentAtom } from '@/lib/persisted' +import { readKey } from '@/lib/storage' -const STATUSBAR_HIDDEN_STORAGE_KEY = 'hermes.desktop.statusbarHidden' +// v1 (`hermes.desktop.statusbarHidden`) was seeded with the approval pill +// hidden, so every existing store carries an `approval-mode` the user never +// chose. v2 seeds from v1 minus that id: other customizations survive, the +// pill appears once on update, and hiding it again persists here. +const STATUSBAR_HIDDEN_STORAGE_KEY = 'hermes.desktop.statusbarHidden.v2' +const LEGACY_HIDDEN_STORAGE_KEY = 'hermes.desktop.statusbarHidden' // v1 (`hermes.desktop.statusbarVisible`) shipped a stretch where the bar was // opt-in, so many stores hold a `false` the user never chose. v2 is read fresh // and the v1 key is deliberately NOT seeded from: every existing install comes @@ -18,14 +24,15 @@ export function toggleStatusbarVisible() { // Items the bar hides until the user turns them on from its context menu. The // bar's job is to answer "is the backend healthy, where am I, what's it doing" — -// route shortcuts (cron/webhooks/agents), the terminal toggle, and the approval -// pill are navigation, not status, so they start out of the way. The per-turn +// route shortcuts (cron/webhooks/agents) and the terminal toggle are +// navigation, not status, so they start out of the way. The approval pill +// (the yolo zap) stays: whether dangerous commands run unasked is state the +// user should see at a glance. The per-turn // session readouts (running/session timers, context meter, cache hit rate, // tokens/sec) are diagnostics most users don't watch, so they start hidden too // and the bar stays quiet mid-turn. export const STATUSBAR_HIDDEN_BY_DEFAULT: readonly string[] = [ 'agents', - 'approval-mode', 'cache-hit-rate', 'context-usage', 'cron', @@ -42,12 +49,27 @@ export const STATUSBAR_HIDDEN_BY_DEFAULT: readonly string[] = [ // staying off. An empty array is a real value — the user turned everything on — // so this uses a sanitizing json codec rather than Codecs.stringArray, which // drops the key when empty and would resurrect the defaults on next launch. +const sanitizeHiddenIds = (value: unknown): string[] => + Array.isArray(value) ? value.filter((id): id is string => typeof id === 'string' && id.length > 0) : [] + +function legacyHiddenSeed(): string[] { + const raw = readKey(LEGACY_HIDDEN_STORAGE_KEY) + + if (raw === null) { + return [...STATUSBAR_HIDDEN_BY_DEFAULT] + } + + try { + return sanitizeHiddenIds(JSON.parse(raw)).filter(id => id !== 'approval-mode') + } catch { + return [...STATUSBAR_HIDDEN_BY_DEFAULT] + } +} + export const $statusbarHiddenIds = persistentAtom( STATUSBAR_HIDDEN_STORAGE_KEY, - [...STATUSBAR_HIDDEN_BY_DEFAULT], - Codecs.json(value => - Array.isArray(value) ? value.filter((id): id is string => typeof id === 'string' && id.length > 0) : [] - ) + legacyHiddenSeed(), + Codecs.json(sanitizeHiddenIds) ) export function setStatusbarItemVisible(id: string, visible: boolean) { From dd1d7475caacfb48ad2a88f60da2f677b5958f4e Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:40:23 +0530 Subject: [PATCH 076/276] chore(contributors): map Collin.Snitchler@pm.me -> ColDSnit (PR #93869 salvage) --- contributors/emails/Collin.Snitchler@pm.me | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/Collin.Snitchler@pm.me diff --git a/contributors/emails/Collin.Snitchler@pm.me b/contributors/emails/Collin.Snitchler@pm.me new file mode 100644 index 0000000000..e467b3b836 --- /dev/null +++ b/contributors/emails/Collin.Snitchler@pm.me @@ -0,0 +1,2 @@ +ColDSnit +# PR #93869 salvage From 63316c74a86328bf0df07251bcef4a24bdf4f0e6 Mon Sep 17 00:00:00 2001 From: ColDSnit Date: Sun, 6 Sep 2026 14:40:39 +0530 Subject: [PATCH 077/276] fix(sessions): compacted display history keeps protected-tail copies in original chronological order _dedupe_display_generations chose the right representative row per logical message but sorted the survivors by that representative's id. A protected-tail copy written into a newer compaction generation has a higher id than messages emitted after the original, so include_compacted reads came back as C, A, B. Anchor the sort on the logical message's first-ever row id instead. Salvaged from #93869 (the tui_gateway half of that PR is superseded by #100504 and #104137); the code moved from hermes_state.py to hermes_state_messages.py since, so the change is re-applied to its new home with the PR's regression test verbatim. --- hermes_state_messages.py | 6 ++++- .../test_get_messages_include_compacted.py | 26 +++++++++++++++++++ 2 files changed, 31 insertions(+), 1 deletion(-) diff --git a/hermes_state_messages.py b/hermes_state_messages.py index 9f09886b5f..07cb4c69ee 100644 --- a/hermes_state_messages.py +++ b/hermes_state_messages.py @@ -588,6 +588,7 @@ class SessionMessagesMixin: into each generation: same role/content/timestamp, different ``active``/id); prefer the live row, then the newest. The ONE definition every display projection shares. *rows* must be ordered by ``id``.""" seen: Dict[Tuple[Any, ...], Any] = {} + first_id: Dict[Tuple[Any, ...], int] = {} for row in rows: dedupe_content = row["content"] if row["role"] == "user": @@ -604,7 +605,10 @@ class SessionMessagesMixin: cur = seen.get(key) if cur is None or (row["active"], row["id"]) > (cur["active"], cur["id"]): seen[key] = row - return sorted(seen.values(), key=lambda r: r["id"]) + first_id[key] = min(first_id.get(key, row["id"]), row["id"]) + # Order by the logical message's FIRST row, not the chosen representative's: a protected-tail + # copy in a newer generation has a higher id than messages emitted after the original. + return [seen[key] for key in sorted(seen, key=first_id.__getitem__)] def _row_to_message_dict(self, row, *, warn_context: str, summary_flag: bool) -> Dict[str, Any]: """``dict(row)`` with content/tool_calls/display_metadata decoded; *summary_flag* keeps diff --git a/tests/hermes_state/test_get_messages_include_compacted.py b/tests/hermes_state/test_get_messages_include_compacted.py index 4b09c7a3ae..c7386ee35a 100644 --- a/tests/hermes_state/test_get_messages_include_compacted.py +++ b/tests/hermes_state/test_get_messages_include_compacted.py @@ -158,6 +158,32 @@ class TestDisplayDedupe: assert len(msgs) == 2 assert [m["content"] for m in msgs] == ["turn 1", "answer 1"] + def test_new_generation_copy_keeps_original_chronological_position(self, db): + """A protected-tail copy inserted after a newer message stays in its + original position in the display read (C, A, B regression).""" + sid = "s1" + db.create_session(sid, source="cli") + db.append_messages_batch( + sid, + [ + {"role": "assistant", "content": "A", "timestamp": 100.0}, + {"role": "assistant", "content": "B", "timestamp": 200.0}, + ], + ) + original = _row_ids(db, sid) + db._execute_write( + lambda conn: conn.execute( + "UPDATE messages SET active = 0, compacted = 1 WHERE session_id = ?", + [sid], + ) + ) + db.append_message(sid, role="user", content="C", timestamp=300.0) + self._copy_tail_as_new_generation(db, sid, original) + + msgs = db.get_messages(sid, include_compacted=True) + + assert [m["content"] for m in msgs] == ["A", "B", "C"] + def test_dedupe_prefers_live_row_then_newest_generation(self, db): """When generations conflict, the live row wins; otherwise the newest generation (highest id) wins.""" From 12871bd01ede1c6eaf3bf456066d1e010034185f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:09:47 -0700 Subject: [PATCH 078/276] feat(openrouter): per-model provider_routing.models. overrides `provider_routing.models.` now takes the same only/ignore/order/sort/ require_parameters/data_collection keys and overlays the flat provider_routing values whenever the agent is on that model. Resolution lives in the one chokepoint every request path already uses (_provider_preferences_for_agent), so CLI, gateway, TUI/Desktop, cron, /model switches, fallback activation and delegated children on another model all honour it with no per-surface plumbing. Matching is spelling-tolerant, sharing _canonical_model_variants with agent.reasoning_overrides. The OpenRouter profile's speed-tier pin no longer overwrites an explicit user `only` on the BASE gpt-6-astra slug: the pin exists to keep default routing off flex/fast, and a user pin is the stronger intent (only: [openai] stays [openai] instead of becoming [openai, azure, azure/us]). Tier slugs (-fast/-flex) keep owning `only`. Live A/B (config only: {gpt-6-astra: [openai], claude-fable-5.1: [anthropic]}): main sent {"sort":"price"} for fable and OpenRouter served it from Azure; with this change it sends {"only":["anthropic"],"sort":"price"} and Anthropic serves it. Schema proposed in #24495 (samplesabotage) and #100711 (Artemonim); this is a slim chokepoint implementation of that design. Co-authored-by: samplesabotage --- agent/chat_completion_helpers.py | 27 ++++++++----- agent/turn_recovery.py | 3 +- cli-config.yaml.example | 8 ++++ hermes_constants.py | 13 ++++++ .../model-providers/openrouter/__init__.py | 5 ++- .../agent/test_per_model_provider_routing.py | 40 +++++++++++++++++++ website/docs/integrations/providers.md | 5 ++- .../user-guide/features/provider-routing.md | 25 ++++++++++++ 8 files changed, 113 insertions(+), 13 deletions(-) create mode 100644 tests/agent/test_per_model_provider_routing.py diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index 98f4f25677..f3952ae1cf 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -378,15 +378,24 @@ def _validated_openrouter_provider_sort(raw_sort: Any) -> Optional[str]: def _provider_preferences_for_agent(agent) -> Dict[str, Any]: - """Build the validated provider-routing object shared by request paths.""" - preferences: Dict[str, Any] = {} - for key, value in (("only", agent.providers_allowed), ("ignore", agent.providers_ignored), - ("order", agent.providers_order), ("sort", _validated_openrouter_provider_sort(agent.provider_sort)), - ("require_parameters", True if agent.provider_require_parameters else None), - ("data_collection", agent.provider_data_collection)): - if value: - preferences[key] = value - return preferences + """Build the validated provider-routing object shared by request paths. + + ``provider_routing.models.`` overlays the flat constructor values for the CURRENT + ``agent.model`` (so ``/model`` switches, fallbacks, and delegated children on another + model each get their own pins without any surface re-plumbing the kwargs).""" + flat = {"only": agent.providers_allowed, "ignore": agent.providers_ignored, "order": agent.providers_order, + "sort": agent.provider_sort, "require_parameters": agent.provider_require_parameters, + "data_collection": agent.provider_data_collection} + per_model = {} + with contextlib.suppress(Exception): + from hermes_cli.config import load_config_readonly + from hermes_constants import resolve_per_model_provider_routing + _pr = load_config_readonly().get("provider_routing") + per_model = resolve_per_model_provider_routing(agent.model, (_pr or {}).get("models") if isinstance(_pr, dict) else None) + merged = {**flat, **{k: v for k, v in per_model.items() if k in flat}} + merged["sort"] = _validated_openrouter_provider_sort(merged["sort"]) + merged["require_parameters"] = True if merged["require_parameters"] else None + return {key: value for key, value in merged.items() if value} def _prompt_cache_scope_for_agent(agent) -> "str | None": diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py index 4dd2f1b5a1..5aed02349a 100644 --- a/agent/turn_recovery.py +++ b/agent/turn_recovery.py @@ -892,7 +892,8 @@ def log_api_error_attempt( if agent._is_openrouter_url() and "support tool use" in error_msg: _blines(agent, f" 💡 No OpenRouter providers for {_model} support tool calling with your current settings.") - if agent.providers_allowed: + from agent.chat_completion_helpers import _provider_preferences_for_agent + if _provider_preferences_for_agent(agent).get("only"): _blines( agent, " Your provider_routing.only restriction is filtering out tool-capable providers.", diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 687badff7d..94ec7e5eda 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -301,6 +301,14 @@ model: # # # Data policy: "allow" (default) or "deny" to exclude providers that may store data # # data_collection: "deny" +# +# # Per-model overrides: same keys, applied only when the agent is on that model +# # (spelling-tolerant match; unset keys fall through to the flat values above). +# # models: +# # "openai/gpt-6-astra": +# # only: ["openai"] +# # "anthropic/claude-fable-5.1": +# # only: ["anthropic"] # ============================================================================= # OpenRouter Response Caching (only applies when using OpenRouter) diff --git a/hermes_constants.py b/hermes_constants.py index 4dc3cd33e7..353764b6ea 100644 --- a/hermes_constants.py +++ b/hermes_constants.py @@ -941,6 +941,19 @@ def resolve_per_model_reasoning_effort(model: str, overrides: dict | None) -> di return None +def resolve_per_model_provider_routing(model: str, models: dict | None) -> dict: + """``provider_routing.models.`` entry for *model*, spelling-tolerant like + ``reasoning_overrides``; ``{}`` when none matches. Only the keys a user sets per model + are returned so unset ones fall through to the flat ``provider_routing`` values.""" + if not model or not isinstance(models, dict): + return {} + for variant in _canonical_model_variants(model): + entry = models.get(variant) + if isinstance(entry, dict): + return entry + return {} + + def resolve_reasoning_config(cfg: dict | None, model: str = "") -> dict | None: """Effective reasoning config for *model*: per-model override, then global ``agent.reasoning_effort``. diff --git a/plugins/model-providers/openrouter/__init__.py b/plugins/model-providers/openrouter/__init__.py index fe654deaba..6e82adfffd 100644 --- a/plugins/model-providers/openrouter/__init__.py +++ b/plugins/model-providers/openrouter/__init__.py @@ -127,8 +127,9 @@ class OpenRouterProfile(ProviderProfile): body["session_id"] = sticky_key prefs = context.get("provider_preferences") pin = OPENROUTER_ENDPOINT_PINS.get(context.get("model") or "") - if pin: - # The tier pin owns ``only``; the user's other routing prefs (ignore/sort/...) still apply. + # The tier pin owns ``only`` (ignore/sort/... still apply) — except on the BASE slug, where the pin + # merely keeps default routing off flex/fast and an explicit user ``only`` is the stronger intent. + if pin and not (pin[0] == context.get("model") and (prefs or {}).get("only")): prefs = {**(prefs or {}), "only": list(pin[1])} if prefs: body["provider"] = prefs diff --git a/tests/agent/test_per_model_provider_routing.py b/tests/agent/test_per_model_provider_routing.py new file mode 100644 index 0000000000..37339f3f33 --- /dev/null +++ b/tests/agent/test_per_model_provider_routing.py @@ -0,0 +1,40 @@ +"""``provider_routing.models.`` overlays the flat OpenRouter routing for the CURRENT agent.model.""" +from types import SimpleNamespace + +import pytest + +from agent import chat_completion_helpers as cch + + +def _agent(model, **flat): + base = dict(providers_allowed=None, providers_ignored=None, providers_order=None, provider_sort="price", + provider_require_parameters=False, provider_data_collection=None) + base.update(flat) + return SimpleNamespace(model=model, **base) + + +@pytest.fixture +def routing_cfg(monkeypatch): + cfg = {"provider_routing": {"sort": "price", "models": { + "openai/gpt-6-astra": {"only": ["openai"]}, + "anthropic/claude-fable-5.1": {"only": ["anthropic"], "sort": "throughput"}, + }}} + import hermes_cli.config as config_mod + monkeypatch.setattr(config_mod, "load_config_readonly", lambda: cfg) + return cfg + + +def test_per_model_entry_overlays_flat_routing_for_that_model_only(routing_cfg): + assert cch._provider_preferences_for_agent(_agent("openai/gpt-6-astra")) == {"only": ["openai"], "sort": "price"} + # A per-model key wins over the flat one; unset keys fall through. + assert cch._provider_preferences_for_agent(_agent("anthropic/claude-fable-5.1")) == { + "only": ["anthropic"], "sort": "throughput"} + # Unlisted model keeps the flat behaviour; no pin leaks across models. + assert cch._provider_preferences_for_agent(_agent("moonshotai/kimi-k2.6")) == {"sort": "price"} + + +def test_per_model_match_is_spelling_tolerant_and_follows_model_switch(routing_cfg): + agent = _agent("openrouter/openai/gpt-6-astra", providers_allowed=["together"]) + assert cch._provider_preferences_for_agent(agent)["only"] == ["openai"] + agent.model = "claude-fable-5-1" + assert cch._provider_preferences_for_agent(agent)["only"] == ["anthropic"] diff --git a/website/docs/integrations/providers.md b/website/docs/integrations/providers.md index d9de727acd..006b04c503 100644 --- a/website/docs/integrations/providers.md +++ b/website/docs/integrations/providers.md @@ -1558,9 +1558,12 @@ provider_routing: # order: ["anthropic", "google"] # Try providers in this order # require_parameters: true # Only use providers that support all request params # data_collection: "deny" # Exclude providers that may store/train on data + # models: # Per-model pins (same keys; unset keys fall through) + # "openai/gpt-6-astra": {only: ["openai"]} + # "anthropic/claude-fable-5.1": {only: ["anthropic"]} ``` -**Shortcuts:** Append `:nitro` to any model name for throughput sorting (e.g., `anthropic/claude-sonnet-4:nitro`), or `:floor` for price sorting. +**Shortcuts:** Append `:nitro` to any model name for throughput sorting (e.g., `anthropic/claude-sonnet-4:nitro`), or `:floor` for price sorting. Per-model details: [Provider Routing](/user-guide/features/provider-routing#per-model-overrides-models). ## OpenRouter Pareto Code Router diff --git a/website/docs/user-guide/features/provider-routing.md b/website/docs/user-guide/features/provider-routing.md index ff8a9ef56c..79e58f7b34 100644 --- a/website/docs/user-guide/features/provider-routing.md +++ b/website/docs/user-guide/features/provider-routing.md @@ -102,6 +102,31 @@ provider_routing: data_collection: "deny" ``` +### Per-model overrides (`models`) + +Pin a different provider set per model. Keys under `models` are model ids; each entry takes the same +`sort` / `only` / `ignore` / `order` / `require_parameters` / `data_collection` keys and overrides the +flat value for that model only. Anything you don't set per model falls through to the flat defaults. + +```yaml +provider_routing: + sort: "price" # applies to every model + models: + "openai/gpt-6-astra": + only: ["openai"] # never let a reseller serve this one + "anthropic/claude-fable-5.1": + only: ["anthropic"] + "moonshotai/kimi-k2.6": + order: ["moonshotai", "together"] + sort: "throughput" +``` + +Matching is spelling-tolerant like `agent.reasoning_overrides` (`claude-fable-5.1` / `claude-fable-5-1`, +with or without the `openrouter/` prefix). The override follows the model the agent is *currently* on, so +`/model` switches, fallback activation, cron jobs, and delegated subagents on another model each get their +own pins. Edit `config.yaml` directly for these keys: model ids contain dots, which `hermes config set` +reads as path separators. + ## Practical Examples ### Optimize for Cost From 0edb6b928ae5479d2f1dd6f87988b6dc8c60da03 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:20:49 -0700 Subject: [PATCH 079/276] fix(delegate): subagents never inherit the 1h prompt-cache tier A delegated child copies the parent's prompt_caching.cache_ttl. The 1h tier is priced for a person who steps away between turns (2x write vs 1.25x for 5m); a subagent calls every few seconds for minutes and is gone, so it paid 2x on every tool result and never collected the retention. Live measurement (Sep 5, 40 concurrent Fable 5.1 children via OpenRouter, 1h markers on the wire): cache writes billed at $20/M against $12.51/M for the same run at 5m, ~60% of a write-dominated bill. _apply_child_cache_ttl runs right after child construction: 1h -> 5m, 5m stays, disabled stays disabled, parent untouched. Tests: unit (markers on the wire drop ttl; parent's 1h layout still differs) and the real spawn path with cache_ttl: 1h in a temp HERMES_HOME. --- tests/tools/test_delegate_child_cache_ttl.py | 64 ++++++++++++++++++++ tools/delegate_tool.py | 10 +++ 2 files changed, 74 insertions(+) create mode 100644 tests/tools/test_delegate_child_cache_ttl.py diff --git a/tests/tools/test_delegate_child_cache_ttl.py b/tests/tools/test_delegate_child_cache_ttl.py new file mode 100644 index 0000000000..9b22921a1a --- /dev/null +++ b/tests/tools/test_delegate_child_cache_ttl.py @@ -0,0 +1,64 @@ +"""Subagents never inherit the 1h prompt-cache tier: with ``prompt_caching.cache_ttl: 1h`` the parent +keeps 1h, the child is built at 5m, and the child's wire markers carry no ``ttl``. A disabled cache +stays disabled (the child is not re-enabled to 5m).""" +from types import SimpleNamespace +from agent.prompt_caching import apply_anthropic_cache_control +from tools.delegate_tool import _apply_child_cache_ttl + + +def _markers(messages): + out = [] + for m in messages: + c = m.get("content") + if isinstance(c, list): + out += [b["cache_control"] for b in c if isinstance(b, dict) and "cache_control" in b] + if "cache_control" in m: + out.append(m["cache_control"]) + return out + + +def test_child_1h_becomes_5m_and_wire_markers_drop_ttl(): + child = SimpleNamespace(_cache_ttl="1h") + _apply_child_cache_ttl(child) + assert child._cache_ttl == "5m" + msgs = [{"role": "system", "content": "sys"}, {"role": "user", "content": "hi"}, + {"role": "assistant", "content": "ok"}, {"role": "user", "content": "go"}] + wire = apply_anthropic_cache_control([dict(m) for m in msgs], cache_ttl=child._cache_ttl) + marks = _markers(wire) + assert marks and all("ttl" not in m for m in marks), marks + # and the parent's 1h layout really is different, so this is not a vacuous check + parent_marks = _markers(apply_anthropic_cache_control([dict(m) for m in msgs], cache_ttl="1h")) + assert any(m.get("ttl") == "1h" for m in parent_marks) + + +def test_disabled_and_5m_are_left_alone(): + for ttl in (None, "5m"): + child = SimpleNamespace(_cache_ttl=ttl) + _apply_child_cache_ttl(child) + assert child._cache_ttl == ttl + + +def test_real_spawn_path_applies_it(tmp_path, monkeypatch): + """Through ``_build_child_agent`` with a real AIAgent: parent configured 1h, child ends at 5m, and the + parent is untouched.""" + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + (tmp_path / "config.yaml").write_text( + "prompt_caching:\n cache_ttl: 1h\nmodel:\n default: anthropic/claude-sonnet-4.6\n", encoding="utf-8") + from run_agent import AIAgent + from tools import delegate_tool as dt + import tools.delegate_tool_config as dtc + monkeypatch.setattr(dt, "_load_config", lambda: {}) + monkeypatch.setattr(dtc, "_load_config", lambda: {}) + kw = dict(api_key="k", base_url="https://openrouter.ai/api/v1", provider="openrouter", + api_mode="chat_completions", model="anthropic/claude-sonnet-4.6", platform="cli", quiet_mode=True, + skip_context_files=True, skip_memory=True, save_trajectories=False, enabled_toolsets=["file"]) + parent = AIAgent(session_id="p", **kw) + assert parent._cache_ttl == "1h", "fixture: the operator's 1h must actually be in effect" + child = dt._build_child_agent(task_index=0, goal="goal", context=None, toolsets=["file"], model=None, + max_iterations=4, task_count=1, parent_agent=parent) + try: + assert child._cache_ttl == "5m" + assert parent._cache_ttl == "1h" + finally: + child.close() + parent.close() diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 70e98fe6a6..9230534e9a 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -102,6 +102,15 @@ def _open_child_session_db(parent_agent) -> Any: return acquire(_parent_db_path) if _parent_db_path is not None else acquire() return None +def _apply_child_cache_ttl(child) -> None: + """A delegated child never uses the 1h cache tier. The tier is priced for a person who steps + away between turns (2x write vs 1.25x for 5m, #14971); a subagent calls every few seconds for + minutes and is gone, so it pays the 2x on every tool result and never collects the retention. + Caching itself stays exactly as configured (disabled stays disabled).""" + if getattr(child, "_cache_ttl", None) == "1h": + child._cache_ttl = "5m" + + def _build_child_agent( task_index: int, goal: str, @@ -195,6 +204,7 @@ def _build_child_agent( release_or_close(child_session_db) raise child._print_fn = getattr(parent_agent, "_print_fn", None) + _apply_child_cache_ttl(child) if child_session_db is not None: child._owns_session_db = True # released by the child's close(), never by the parent # Ownership transfer for the dedicated handle: the child's close() must release it (nothing else holds a From c3718d5500cd517de463c54200a95858d531c0a2 Mon Sep 17 00:00:00 2001 From: Halldrix <12357213+Halldrix@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:46:30 -0500 Subject: [PATCH 080/276] fix(memory): refuse batch that would empty a non-empty store (#103419) apply_batch() committed an empty USER.md/MEMORY.md as a normal successful write when a consolidation batch removed the last entry. Refuse all-or-nothing with live entries so background consolidation keeps at least one entry; a deliberate wipe stays a manual file edit. --- tools/memory_tool_store.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/tools/memory_tool_store.py b/tools/memory_tool_store.py index dcb04ef364..4dcc3b5701 100644 --- a/tools/memory_tool_store.py +++ b/tools/memory_tool_store.py @@ -317,6 +317,19 @@ class MemoryStore: (op.get("old_text") or "").strip(), f"Operation {i + 1} ({act or 'unknown'})") if msg: return self._failure_with_entries(target, msg + " No operations were applied (batch is all-or-nothing).") + if entries and not working: + # A batch must never silently empty a previously non-empty + # store (#103419): a background consolidation that removes the + # last entry would otherwise commit an empty USER.md/MEMORY.md + # as a normal successful write. Refuse all-or-nothing so the + # model keeps at least one entry; a deliberate wipe is a manual + # file edit, not a consolidation side effect. + label = "USER.md" if target == "user" else "MEMORY.md" + return self._failure_with_entries(target, ( + f"Refusing to empty {label}: this batch would remove every entry from a " + f"previously non-empty profile. Nothing was applied (batch is all-or-nothing). " + f"Keep at least one entry — merge overlapping entries into a shorter one instead " + f"of removing the last one (see current_entries below).")) new_total = len(ENTRY_DELIMITER.join(working)) # budget check against the FINAL state only if new_total > limit: return self._failure_with_entries(target, ( From faa46ed2c7ac65ad62280f15f3464643976f24f4 Mon Sep 17 00:00:00 2001 From: Halldrix <12357213+Halldrix@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:46:30 -0500 Subject: [PATCH 081/276] test(memory): pin batch refusal to empty a non-empty store (#103419) --- tests/tools/test_memory_tool.py | 52 +++++++++++++++++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/tests/tools/test_memory_tool.py b/tests/tools/test_memory_tool.py index 2af6107bfc..6b7346ccf2 100644 --- a/tests/tools/test_memory_tool.py +++ b/tests/tools/test_memory_tool.py @@ -704,3 +704,55 @@ class TestBomToleranceInMemoryFiles: raw, read_ok = MemoryStore._read_raw_checked(path) assert read_ok is False assert raw == "" + + +# ========================================================================= +# Batch must not silently empty a non-empty store (#103419) +# +# A background consolidation batch that removes the last remaining entry +# commits an empty USER.md/MEMORY.md as a normal successful write — silent +# profile data loss. The batch path must refuse to reduce a previously +# non-empty target to zero entries (all-or-nothing: nothing is written). +# Single remove-last stays allowed (explicit delete, pinned elsewhere). +# ========================================================================= + + +class TestBatchRefusesToEmptyNonEmptyStore: + @pytest.mark.parametrize( + ("target", "seed"), + [("user", "Name: Alice"), ("memory", "fact A")], + ) + def test_batch_removing_last_entry_is_refused_and_preserves_disk( + self, store, target, seed + ): + assert store.add(target, seed)["success"] is True + path = store._path_for(target) + before = path.read_text(encoding="utf-8") + + result = store.apply_batch(target, [{"action": "remove", "old_text": seed}]) + + assert result["success"] is False + assert "current_entries" in result # actionable, counts toward degrade budget + assert seed in path.read_text(encoding="utf-8") # nothing written + assert path.read_text(encoding="utf-8") == before + assert seed in store._entries_for(target) + + @pytest.mark.parametrize( + ("target", "seed"), + [("user", "Name: Alice"), ("memory", "fact A")], + ) + def test_batch_ending_nonempty_still_succeeds(self, store, target, seed): + assert store.add(target, seed)["success"] is True + assert store.add(target, "second entry here")["success"] is True + + result = store.apply_batch( + target, + [ + {"action": "remove", "old_text": seed}, + {"action": "add", "content": "replacement entry here"}, + ], + ) + + assert result["success"] is True + assert seed not in store._entries_for(target) + assert "replacement entry here" in store._entries_for(target) From 7602b33d79d97d463ee9b0d878425a0deb24e130 Mon Sep 17 00:00:00 2001 From: Halldrix <12357213+Halldrix@users.noreply.github.com> Date: Fri, 4 Sep 2026 23:56:42 -0500 Subject: [PATCH 082/276] fix(memory): point empty-batch refusal at single remove() calls (#103419) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review feedback: the sanctioned deliberate wipe is repeated single-op remove() calls, not a manual file edit — direct the model there. --- tests/tools/test_memory_tool.py | 1 + tools/memory_tool_store.py | 8 +++++--- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/tests/tools/test_memory_tool.py b/tests/tools/test_memory_tool.py index 6b7346ccf2..f1823f0523 100644 --- a/tests/tools/test_memory_tool.py +++ b/tests/tools/test_memory_tool.py @@ -733,6 +733,7 @@ class TestBatchRefusesToEmptyNonEmptyStore: assert result["success"] is False assert "current_entries" in result # actionable, counts toward degrade budget + assert "single remove()" in result["error"] # points at the deliberate-wipe path assert seed in path.read_text(encoding="utf-8") # nothing written assert path.read_text(encoding="utf-8") == before assert seed in store._entries_for(target) diff --git a/tools/memory_tool_store.py b/tools/memory_tool_store.py index 4dcc3b5701..50d8ae141e 100644 --- a/tools/memory_tool_store.py +++ b/tools/memory_tool_store.py @@ -322,14 +322,16 @@ class MemoryStore: # store (#103419): a background consolidation that removes the # last entry would otherwise commit an empty USER.md/MEMORY.md # as a normal successful write. Refuse all-or-nothing so the - # model keeps at least one entry; a deliberate wipe is a manual - # file edit, not a consolidation side effect. + # model keeps at least one entry; deleting the final entry on + # purpose is what single remove() calls are for, not a + # consolidation side effect. label = "USER.md" if target == "user" else "MEMORY.md" return self._failure_with_entries(target, ( f"Refusing to empty {label}: this batch would remove every entry from a " f"previously non-empty profile. Nothing was applied (batch is all-or-nothing). " f"Keep at least one entry — merge overlapping entries into a shorter one instead " - f"of removing the last one (see current_entries below).")) + f"of removing the last one (see current_entries below). To delete the final entry " + f"deliberately, use single remove() calls.")) new_total = len(ENTRY_DELIMITER.join(working)) # budget check against the FINAL state only if new_total > limit: return self._failure_with_entries(target, ( From 117dc02870c79f7281023c49eb83647322622acb Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:25:30 +0530 Subject: [PATCH 083/276] refactor(memory): derive the store label from _path_for, trim the guard comment MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-up on the salvaged #103421 guard: the USER.md/MEMORY.md ternary was a third copy (also _path_for and memory_tool._memory_target_error); use the path's name. "profile" → "store" in the error since MEMORY.md is not a profile. --- tools/memory_tool_store.py | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/tools/memory_tool_store.py b/tools/memory_tool_store.py index 50d8ae141e..4575944443 100644 --- a/tools/memory_tool_store.py +++ b/tools/memory_tool_store.py @@ -318,17 +318,13 @@ class MemoryStore: if msg: return self._failure_with_entries(target, msg + " No operations were applied (batch is all-or-nothing).") if entries and not working: - # A batch must never silently empty a previously non-empty - # store (#103419): a background consolidation that removes the - # last entry would otherwise commit an empty USER.md/MEMORY.md - # as a normal successful write. Refuse all-or-nothing so the - # model keeps at least one entry; deleting the final entry on - # purpose is what single remove() calls are for, not a - # consolidation side effect. - label = "USER.md" if target == "user" else "MEMORY.md" + # #103419: a consolidation batch that removes the last entry would + # commit an empty file as a normal successful write. Refuse; single + # remove() is the deliberate-wipe path. + label = self._path_for(target).name return self._failure_with_entries(target, ( f"Refusing to empty {label}: this batch would remove every entry from a " - f"previously non-empty profile. Nothing was applied (batch is all-or-nothing). " + f"previously non-empty store. Nothing was applied (batch is all-or-nothing). " f"Keep at least one entry — merge overlapping entries into a shorter one instead " f"of removing the last one (see current_entries below). To delete the final entry " f"deliberately, use single remove() calls.")) From 10e8756cf4d684d569c834f1562138b1e77423c4 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:32:20 +0530 Subject: [PATCH 084/276] test(memory): drop assertions implied by the disk-unchanged check --- tests/tools/test_memory_tool.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/tests/tools/test_memory_tool.py b/tests/tools/test_memory_tool.py index f1823f0523..44d43d4d31 100644 --- a/tests/tools/test_memory_tool.py +++ b/tests/tools/test_memory_tool.py @@ -733,10 +733,8 @@ class TestBatchRefusesToEmptyNonEmptyStore: assert result["success"] is False assert "current_entries" in result # actionable, counts toward degrade budget - assert "single remove()" in result["error"] # points at the deliberate-wipe path - assert seed in path.read_text(encoding="utf-8") # nothing written - assert path.read_text(encoding="utf-8") == before - assert seed in store._entries_for(target) + assert "remove" in result["error"] # points at the deliberate-wipe path + assert path.read_text(encoding="utf-8") == before # nothing written @pytest.mark.parametrize( ("target", "seed"), From cd27df3c2ec3c1cd5867b1f500bab3b6f1b6ae9c Mon Sep 17 00:00:00 2001 From: tachyon-r <291518778+tachyon-r@users.noreply.github.com> Date: Fri, 28 Aug 2026 20:46:12 -0400 Subject: [PATCH 085/276] fix(update): ignore successful receipt stop reason --- hermes_cli/update_cmd_fleet.py | 27 +++++--- .../test_update_fleet_restart_pending.py | 61 +++++++++++++++++++ 2 files changed, 81 insertions(+), 7 deletions(-) diff --git a/hermes_cli/update_cmd_fleet.py b/hermes_cli/update_cmd_fleet.py index e9f1c92875..b8ebab13f3 100644 --- a/hermes_cli/update_cmd_fleet.py +++ b/hermes_cli/update_cmd_fleet.py @@ -78,14 +78,27 @@ def _current_checkout_sha() -> str | None: def _receipt_looks_unfinished(receipt: dict) -> bool: - """True when *receipt* is from an update that did not finish cleanly.""" + """True when *receipt* is from an update that did not finish cleanly. + + ``stop_reason`` records *how* the command boundary closed the receipt + (``completed at command boundary``, ``sys.exit(0)``, even KeyboardInterrupt + on an otherwise successful run). A truthy stop_reason must not make a + successful receipt look unfinished, or the next ``hermes update`` retriggers + ``fleet_restart_pending`` from pre-pull plan SHAs. + """ + exit_code = receipt.get("exit_code") + outcome = receipt.get("outcome") + if exit_code not in (0, None): + return True + if outcome in ("failed", "partial", "running"): + return True gateway_restart = receipt.get("gateway_restart") - return bool( - receipt.get("stop_reason") - or receipt.get("exit_code") not in (0, None) - or receipt.get("outcome") in ("failed", "partial", "running") - or (isinstance(gateway_restart, dict) and gateway_restart.get("incomplete")) - ) + if isinstance(gateway_restart, dict) and gateway_restart.get("incomplete"): + return True + stop_reason = receipt.get("stop_reason") + if stop_reason and outcome != "success" and exit_code != 0: + return True + return False def _receipt_reports_stale_runtime(expected_sha: str | None = None) -> bool: diff --git a/tests/hermes_cli/test_update_fleet_restart_pending.py b/tests/hermes_cli/test_update_fleet_restart_pending.py index 156573a92e..49fcf3705b 100644 --- a/tests/hermes_cli/test_update_fleet_restart_pending.py +++ b/tests/hermes_cli/test_update_fleet_restart_pending.py @@ -254,6 +254,67 @@ def test_successful_receipt_with_pre_update_plan_shas_does_not_retrigger( assert update_cmd._pending_fleet_restart_needed() is False +def test_successful_command_boundary_receipt_without_fleet_does_not_retrigger( + monkeypatch, +): + """A normal command-boundary stop is not an interrupted update.""" + disk_sha = "n" * 40 + old_sha = "o" * 40 + monkeypatch.setattr(update_cmd, "_current_checkout_sha", lambda: disk_sha) + monkeypatch.setattr(update_cmd_fleet, "_current_checkout_sha", lambda: disk_sha) + + receipt_dir = get_hermes_home() / "logs" / "update_receipts" + receipt_dir.mkdir(parents=True) + (receipt_dir / "latest.json").write_text( + json.dumps( + { + "exit_code": 0, + "outcome": "success", + "stop_reason": "completed at command boundary", + "plan": { + "expected_sha": old_sha, + "runtimes": [ + { + "kind": "gateway", + "profile": "default", + "pid": 1, + "code_sha": old_sha, + } + ], + }, + "fleet": [], + "gateway_restart": {}, + } + ), + encoding="utf-8", + ) + + assert update_cmd._pending_fleet_restart_needed() is False + + +@pytest.mark.parametrize( + "receipt", + [ + {"outcome": "success", "exit_code": 0, "stop_reason": "sys.exit(0)"}, + {"outcome": "success", "stop_reason": "KeyboardInterrupt: "}, + {"exit_code": 0, "stop_reason": "sys.exit(0)"}, + ], +) +def test_successful_non_boundary_stop_reasons_are_finished(receipt): + """Successful sys.exit(0)/KeyboardInterrupt must not look unfinished.""" + assert update_cmd._receipt_looks_unfinished(receipt) is False + + +def test_failed_interrupt_stop_reason_is_unfinished(): + assert update_cmd._receipt_looks_unfinished( + { + "outcome": "failed", + "exit_code": 1, + "stop_reason": "KeyboardInterrupt: ", + } + ) is True + + def test_stale_fleet_matrix_on_latest_receipt_is_pending(monkeypatch): disk_sha = "n" * 40 monkeypatch.setattr(update_cmd, "_current_checkout_sha", lambda: disk_sha) From 6e0fba4e58985422feddd5f5d992c0e34b50850b Mon Sep 17 00:00:00 2001 From: tachyon-r <291518778+tachyon-r@users.noreply.github.com> Date: Sun, 30 Aug 2026 13:18:00 -0400 Subject: [PATCH 086/276] refactor(update): share command-boundary receipt reason --- hermes_cli/main.py | 4 +++- hermes_cli/update_receipt.py | 1 + tests/hermes_cli/test_update_fleet_restart_pending.py | 3 ++- 3 files changed, 6 insertions(+), 2 deletions(-) diff --git a/hermes_cli/main.py b/hermes_cli/main.py index 6cb431dae7..8c73325067 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -2279,7 +2279,9 @@ def cmd_update(args): _finalize_update_receipt(1, f"{type(_update_exc).__name__}: {_update_exc}") raise else: - _finalize_update_receipt(0, "completed at command boundary") + from hermes_cli.update_receipt import COMMAND_BOUNDARY_STOP_REASON + + _finalize_update_receipt(0, COMMAND_BOUNDARY_STOP_REASON) _update_handoff_exit_code = 0 finally: _update_lock.release() diff --git a/hermes_cli/update_receipt.py b/hermes_cli/update_receipt.py index 1a4fecabab..85135e712a 100644 --- a/hermes_cli/update_receipt.py +++ b/hermes_cli/update_receipt.py @@ -19,6 +19,7 @@ from typing import Any, Optional logger = logging.getLogger(__name__) _RECEIPT_KEEP = 20 # keep the last N receipts per profile home +COMMAND_BOUNDARY_STOP_REASON = "completed at command boundary" # ``hermes update`` is a single-threaded CLI command; a module singleton lets the 7k-line updater # record steps from any depth without threading a handle through every helper. diff --git a/tests/hermes_cli/test_update_fleet_restart_pending.py b/tests/hermes_cli/test_update_fleet_restart_pending.py index 49fcf3705b..c099f992c9 100644 --- a/tests/hermes_cli/test_update_fleet_restart_pending.py +++ b/tests/hermes_cli/test_update_fleet_restart_pending.py @@ -27,6 +27,7 @@ import hermes_cli.main_install_repair as main_install_repair from hermes_cli import update_cmd import hermes_cli.update_cmd_fleet as update_cmd_fleet import hermes_cli.update_cmd_deps as update_cmd_deps +from hermes_cli.update_receipt import COMMAND_BOUNDARY_STOP_REASON from hermes_constants import get_hermes_home @@ -270,7 +271,7 @@ def test_successful_command_boundary_receipt_without_fleet_does_not_retrigger( { "exit_code": 0, "outcome": "success", - "stop_reason": "completed at command boundary", + "stop_reason": COMMAND_BOUNDARY_STOP_REASON, "plan": { "expected_sha": old_sha, "runtimes": [ From f7e56d4256adeb6c914ce9038c2791d00308fdba Mon Sep 17 00:00:00 2001 From: tachyon-r <291518778+tachyon-r@users.noreply.github.com> Date: Sun, 30 Aug 2026 13:24:30 -0400 Subject: [PATCH 087/276] test(update): isolate fleet restart tests from launchd --- tests/hermes_cli/test_update_fleet_restart_pending.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/tests/hermes_cli/test_update_fleet_restart_pending.py b/tests/hermes_cli/test_update_fleet_restart_pending.py index c099f992c9..ccde52ffc7 100644 --- a/tests/hermes_cli/test_update_fleet_restart_pending.py +++ b/tests/hermes_cli/test_update_fleet_restart_pending.py @@ -83,6 +83,12 @@ def _patch_update_deps(monkeypatch, tmp_path, run_side_effect): monkeypatch.setattr(hermes_main, "_resolve_update_branch", lambda args: "main") monkeypatch.setattr(hermes_main, "_is_windows", lambda: False) monkeypatch.setattr(main_install_repair, "_is_windows", lambda: False) + monkeypatch.setattr( + update_cmd, "_restart_macos_launchd_gateways", lambda *a, **k: None + ) + monkeypatch.setattr( + update_cmd_fleet, "_restart_macos_launchd_gateways", lambda *a, **k: None + ) monkeypatch.setattr( hermes_main, "_get_origin_url", From dad698d88cc4c6824d18f7ae9b7c64cbbf54643e Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:25:31 +0530 Subject: [PATCH 088/276] test(update): pin the refused-receipt shape the stop_reason clause exists for update_contract writes {"outcome": "refused", "stop_reason": } with no exit_code; that is the one production receipt where the stop_reason clause in _receipt_looks_unfinished is load-bearing. The previous negative control used exit_code=1, which the exit_code branch already catches. Docstring reworded: a KeyboardInterrupt never lands on a success receipt (the boundary finalize is a no-op once the inner path finalized). --- hermes_cli/update_cmd_fleet.py | 4 ++-- tests/hermes_cli/test_update_fleet_restart_pending.py | 8 ++++++++ 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/hermes_cli/update_cmd_fleet.py b/hermes_cli/update_cmd_fleet.py index b8ebab13f3..8aa42deeeb 100644 --- a/hermes_cli/update_cmd_fleet.py +++ b/hermes_cli/update_cmd_fleet.py @@ -81,8 +81,8 @@ def _receipt_looks_unfinished(receipt: dict) -> bool: """True when *receipt* is from an update that did not finish cleanly. ``stop_reason`` records *how* the command boundary closed the receipt - (``completed at command boundary``, ``sys.exit(0)``, even KeyboardInterrupt - on an otherwise successful run). A truthy stop_reason must not make a + (``completed at command boundary``, ``sys.exit(0)``, or a refusal code from + ``update_contract``). A truthy stop_reason must not make a successful receipt look unfinished, or the next ``hermes update`` retriggers ``fleet_restart_pending`` from pre-pull plan SHAs. """ diff --git a/tests/hermes_cli/test_update_fleet_restart_pending.py b/tests/hermes_cli/test_update_fleet_restart_pending.py index ccde52ffc7..321fa25bf8 100644 --- a/tests/hermes_cli/test_update_fleet_restart_pending.py +++ b/tests/hermes_cli/test_update_fleet_restart_pending.py @@ -312,6 +312,14 @@ def test_successful_non_boundary_stop_reasons_are_finished(receipt): assert update_cmd._receipt_looks_unfinished(receipt) is False +def test_refused_receipt_with_only_a_stop_reason_is_unfinished(): + # update_contract writes {"outcome": "refused", "stop_reason": } with no exit_code; + # the stop_reason clause is what keeps that receipt unfinished. + assert update_cmd._receipt_looks_unfinished( + {"outcome": "refused", "stop_reason": "not_updatable_in_place"} + ) is True + + def test_failed_interrupt_stop_reason_is_unfinished(): assert update_cmd._receipt_looks_unfinished( { From bc1330eebc0aa8a443501b5f62586eb7361353a5 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:32:23 +0530 Subject: [PATCH 089/276] refactor(update): name the success invariant in _receipt_looks_unfinished; one predicate contract The tail clause was correct only by ordering (exit_code != 0 there meant exit_code is None). Name what it encodes: a stop_reason counts only when nothing vouched for success. Same truth table. The three literal-dict tests on the predicate collapse into one parametrized contract; the handoff-exit test binds to COMMAND_BOUNDARY_STOP_REASON instead of re-spelling it. --- hermes_cli/update_cmd_fleet.py | 19 +++++----- .../test_update_fleet_restart_pending.py | 35 ++++++------------- tests/hermes_cli/test_update_handoff_exit.py | 5 +-- 3 files changed, 21 insertions(+), 38 deletions(-) diff --git a/hermes_cli/update_cmd_fleet.py b/hermes_cli/update_cmd_fleet.py index 8aa42deeeb..d637db9556 100644 --- a/hermes_cli/update_cmd_fleet.py +++ b/hermes_cli/update_cmd_fleet.py @@ -80,25 +80,22 @@ def _current_checkout_sha() -> str | None: def _receipt_looks_unfinished(receipt: dict) -> bool: """True when *receipt* is from an update that did not finish cleanly. - ``stop_reason`` records *how* the command boundary closed the receipt - (``completed at command boundary``, ``sys.exit(0)``, or a refusal code from - ``update_contract``). A truthy stop_reason must not make a + The command boundary stamps a ``stop_reason`` on every receipt, including clean + ones (``completed at command boundary``, ``sys.exit(0)``); it must not make a successful receipt look unfinished, or the next ``hermes update`` retriggers - ``fleet_restart_pending`` from pre-pull plan SHAs. + ``fleet_restart_pending`` from pre-pull plan SHAs (#98022). """ exit_code = receipt.get("exit_code") outcome = receipt.get("outcome") - if exit_code not in (0, None): - return True - if outcome in ("failed", "partial", "running"): + if exit_code not in (0, None) or outcome in ("failed", "partial", "running"): return True gateway_restart = receipt.get("gateway_restart") if isinstance(gateway_restart, dict) and gateway_restart.get("incomplete"): return True - stop_reason = receipt.get("stop_reason") - if stop_reason and outcome != "success" and exit_code != 0: - return True - return False + # A stop_reason alone (update_contract refusals: outcome="refused", no exit_code) + # counts only when nothing else vouched for success. + succeeded = exit_code == 0 or outcome == "success" + return bool(receipt.get("stop_reason")) and not succeeded def _receipt_reports_stale_runtime(expected_sha: str | None = None) -> bool: diff --git a/tests/hermes_cli/test_update_fleet_restart_pending.py b/tests/hermes_cli/test_update_fleet_restart_pending.py index 321fa25bf8..4056d73bc6 100644 --- a/tests/hermes_cli/test_update_fleet_restart_pending.py +++ b/tests/hermes_cli/test_update_fleet_restart_pending.py @@ -300,34 +300,19 @@ def test_successful_command_boundary_receipt_without_fleet_does_not_retrigger( @pytest.mark.parametrize( - "receipt", + ("receipt", "unfinished"), [ - {"outcome": "success", "exit_code": 0, "stop_reason": "sys.exit(0)"}, - {"outcome": "success", "stop_reason": "KeyboardInterrupt: "}, - {"exit_code": 0, "stop_reason": "sys.exit(0)"}, + pytest.param({"outcome": "success", "exit_code": 0, "stop_reason": "sys.exit(0)"}, False, id="success-sys-exit-0"), + pytest.param({"outcome": "success", "stop_reason": "KeyboardInterrupt: "}, False, id="success-no-exit-code"), + pytest.param({"exit_code": 0, "stop_reason": "sys.exit(0)"}, False, id="exit-0-no-outcome"), + # update_contract writes {"outcome": "refused", "stop_reason": } with no exit_code; + # the stop_reason clause is what keeps that receipt unfinished. + pytest.param({"outcome": "refused", "stop_reason": "not_updatable_in_place"}, True, id="refused-stop-reason-only"), + pytest.param({"outcome": "failed", "exit_code": 1, "stop_reason": "KeyboardInterrupt: "}, True, id="failed-interrupt"), ], ) -def test_successful_non_boundary_stop_reasons_are_finished(receipt): - """Successful sys.exit(0)/KeyboardInterrupt must not look unfinished.""" - assert update_cmd._receipt_looks_unfinished(receipt) is False - - -def test_refused_receipt_with_only_a_stop_reason_is_unfinished(): - # update_contract writes {"outcome": "refused", "stop_reason": } with no exit_code; - # the stop_reason clause is what keeps that receipt unfinished. - assert update_cmd._receipt_looks_unfinished( - {"outcome": "refused", "stop_reason": "not_updatable_in_place"} - ) is True - - -def test_failed_interrupt_stop_reason_is_unfinished(): - assert update_cmd._receipt_looks_unfinished( - { - "outcome": "failed", - "exit_code": 1, - "stop_reason": "KeyboardInterrupt: ", - } - ) is True +def test_stop_reason_only_marks_unfinished_when_nothing_vouches_for_success(receipt, unfinished): + assert update_cmd._receipt_looks_unfinished(receipt) is unfinished def test_stale_fleet_matrix_on_latest_receipt_is_pending(monkeypatch): diff --git a/tests/hermes_cli/test_update_handoff_exit.py b/tests/hermes_cli/test_update_handoff_exit.py index 4b963c1ba8..9d79986c7e 100644 --- a/tests/hermes_cli/test_update_handoff_exit.py +++ b/tests/hermes_cli/test_update_handoff_exit.py @@ -23,6 +23,7 @@ import pytest import hermes_cli.main as main_mod from hermes_cli import update_cmd from hermes_cli.main import cmd_update +from hermes_cli.update_receipt import COMMAND_BOUNDARY_STOP_REASON class _FakeLock: @@ -89,7 +90,7 @@ def _noop_impl(args, gateway_mode=False): def test_handoff_child_hard_exits_zero_after_success(monkeypatch): events = _run_cmd_update(monkeypatch, _noop_impl, reexec=True) assert events["exit_codes"] == [0] - assert events["receipts"] == [(0, "completed at command boundary")] + assert events["receipts"] == [(0, COMMAND_BOUNDARY_STOP_REASON)] # The hard exit is the last thing, after lock release and stdio restore. assert events["order"] == ["acquire", "impl", "release", "restore-stdio", "hard-exit"] @@ -98,7 +99,7 @@ def test_non_handoff_run_never_hard_exits(monkeypatch): events = _run_cmd_update(monkeypatch, _noop_impl, reexec=False) assert events["exit_codes"] == [] assert "hard-exit" not in events["order"] - assert events["receipts"] == [(0, "completed at command boundary")] + assert events["receipts"] == [(0, COMMAND_BOUNDARY_STOP_REASON)] def test_handoff_child_propagates_early_systemexit_code(monkeypatch): From 611ee856c53f9beee52f807c9134b232e4e83977 Mon Sep 17 00:00:00 2001 From: gkd2323c Date: Fri, 4 Sep 2026 01:22:21 +0800 Subject: [PATCH 090/276] fix(process): drop OOMPolicy from systemd-run --scope argv OOMPolicy is a service-unit property; systemd-run --user --scope rejects it with 'Unknown assignment: OOMPolicy=kill'. The availability probe therefore always failed in supervised Linux gateways, making restart-safe cron worker dispatch (systemd scope) permanently unavailable and falling back to in-cgroup workers. MemoryMax + MemoryAccounting remain; OOMPolicy adds nothing for a transient scope (no service manager to act on the OOM event). --- tools/process_registry.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tools/process_registry.py b/tools/process_registry.py index afdd52a0e7..474f97e39e 100644 --- a/tools/process_registry.py +++ b/tools/process_registry.py @@ -131,7 +131,6 @@ def _systemd_scope_argv(binary: str, unit_name: str, *argv: str) -> List[str]: binary, "--user", "--scope", "--quiet", "--unit", unit_name, "--collect", "--property", "MemoryAccounting=yes", "--property", f"MemoryMax={_worker_memory_max_bytes()}", - "--property", "OOMPolicy=kill", "--", *argv, ] From 04885c25cfb61bfa73912eec71b9baaaf4135920 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:07:50 +0530 Subject: [PATCH 091/276] test(process_registry): pin that scope argv never emits OOMPolicy The spawn_local systemd test asserted the property was present; it now asserts the invariant the fix establishes (no OOMPolicy= on a transient scope), so a reintroduction fails here instead of on older systemd hosts. --- tests/tools/test_process_registry.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/tools/test_process_registry.py b/tests/tools/test_process_registry.py index 2c09bc1413..51f86f83e0 100644 --- a/tests/tools/test_process_registry.py +++ b/tests/tools/test_process_registry.py @@ -1956,7 +1956,10 @@ class TestSystemdCgroupIsolation: if value == "--property" ] assert "MemoryAccounting=yes" in properties - assert "OOMPolicy=kill" in properties + # systemd rejects OOMPolicy= on transient --scope units across the versions + # users run (239/245/249, #102486); emitting it fails the probe and every + # cron worker dispatch. MemoryMax + MemoryAccounting carry the isolation. + assert not any(p.startswith("OOMPolicy=") for p in properties), properties memory_max = next( value for value in properties if value.startswith("MemoryMax=") ) From 78fc943ca4e80320607e0bb4e18a268338f886b8 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:25:32 +0530 Subject: [PATCH 092/276] docs(process_registry): say in the argv helper why OOMPolicy is absent --- tools/process_registry.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tools/process_registry.py b/tools/process_registry.py index 474f97e39e..2df6fe1eba 100644 --- a/tools/process_registry.py +++ b/tools/process_registry.py @@ -126,7 +126,8 @@ def _worker_memory_max_bytes() -> int: def _systemd_scope_argv(binary: str, unit_name: str, *argv: str) -> List[str]: """``systemd-run --user --scope`` argv shared by the probe and real spawns. - ``--collect`` self-cleans the scope after exit; ``--unit`` names it for systemctl.""" + ``--collect`` self-cleans the scope after exit; ``--unit`` names it for systemctl. + No ``OOMPolicy=``: transient scopes reject it on systemd <253 (#102486).""" return [ binary, "--user", "--scope", "--quiet", "--unit", unit_name, "--collect", "--property", "MemoryAccounting=yes", From 3513a3b9227f16d46a62ba335fe77e500252980c Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 14:52:28 +0530 Subject: [PATCH 093/276] test(process_registry): pin that the availability probe argv omits OOMPolicy too Requested in the review of #102357: the spawn builder was pinned, the probe was not, and the probe is where the rejection was cached as "unavailable". --- tests/tools/test_process_registry.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/tests/tools/test_process_registry.py b/tests/tools/test_process_registry.py index 51f86f83e0..3ac709db9d 100644 --- a/tests/tools/test_process_registry.py +++ b/tests/tools/test_process_registry.py @@ -2368,6 +2368,12 @@ class TestSystemdCgroupIsolation: assert first is True assert second is True assert len(probe_calls) == 1, "probe must run only once (cached)" + # The probe must not carry OOMPolicy= either: that is the argv systemd + # rejected on scope units and cached as "unavailable" (#102486). + probe_argv = probe_calls[0][0] + assert not any( + value.startswith("OOMPolicy=") for value in probe_argv if isinstance(value, str) + ), probe_argv def test_systemd_scope_first_probe_is_serialized(self, monkeypatch): """Concurrent first-use callers must wait for one definitive probe. From ba7d2a8633468b242bd95994c472ded4759571df Mon Sep 17 00:00:00 2001 From: webtecnica Date: Mon, 3 Aug 2026 08:12:35 -0300 Subject: [PATCH 094/276] fix(providers): stop forwarding provider_routing prefs to Nous Portal --- plugins/model-providers/nous/__init__.py | 7 ++++--- tests/providers/test_provider_profiles.py | 9 +++++++++ 2 files changed, 13 insertions(+), 3 deletions(-) diff --git a/plugins/model-providers/nous/__init__.py b/plugins/model-providers/nous/__init__.py index 1c28cef37f..2e37e3d4a3 100644 --- a/plugins/model-providers/nous/__init__.py +++ b/plugins/model-providers/nous/__init__.py @@ -29,9 +29,10 @@ class NousProfile(ProviderProfile): sticky_key = _cache_scope_from_session_id(get_affinity_scope() or get_conversation_context() or session_id) if sticky_key: body["session_id"] = sticky_key - provider_preferences = context.get("provider_preferences") - if provider_preferences: - body["provider"] = provider_preferences + # Nous Portal inference rejects caller-supplied provider routing prefs + # (only/ignore/order/sort/data_collection/zdr/require_parameters) with + # HTTP 400 — routing is decided centrally per model. provider_routing + # from config.yaml is OpenRouter-only, so it is not forwarded here. return body @staticmethod diff --git a/tests/providers/test_provider_profiles.py b/tests/providers/test_provider_profiles.py index e28445abde..1507dbdf85 100644 --- a/tests/providers/test_provider_profiles.py +++ b/tests/providers/test_provider_profiles.py @@ -225,6 +225,15 @@ class TestNousProfile: assert first["session_id"] == "cron_job42" assert first["session_id"] == second["session_id"] + def test_extra_body_ignores_provider_preferences(self): + """Nous Portal rejects caller-supplied provider routing prefs (HTTP 400).""" + p = get_provider_profile("nous") + body = p.build_extra_body( + provider_preferences={"allow": ["anthropic"], "sort": "price"} + ) + assert "provider" not in body + assert "tags" in body + From 0a195aa4636494812a52ab7068a5ab822d050abf Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:33:42 -0700 Subject: [PATCH 095/276] docs: provider_routing is OpenRouter-only; Nous Portal rejects the provider object Portal decides routing centrally per model and returns HTTP 400 on any caller-supplied `provider` object (live-confirmed). Drop the five "OpenRouter or Nous Portal" claims from the provider-routing page and say plainly that the setting is ignored on Portal now that the profile no longer forwards it. --- .../docs/user-guide/features/provider-routing.md | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/website/docs/user-guide/features/provider-routing.md b/website/docs/user-guide/features/provider-routing.md index 79e58f7b34..3cd2c5c75c 100644 --- a/website/docs/user-guide/features/provider-routing.md +++ b/website/docs/user-guide/features/provider-routing.md @@ -1,18 +1,18 @@ --- title: Provider Routing -description: Configure OpenRouter or Nous Portal provider preferences to optimize for cost, speed, or quality. +description: Configure OpenRouter provider preferences to optimize for cost, speed, or quality. sidebar_label: Provider Routing sidebar_position: 7 --- # Provider Routing -When using [OpenRouter](https://openrouter.ai) or [Nous Portal](/integrations/nous-portal) as your LLM provider, Hermes Agent supports **provider routing** — fine-grained control over which underlying AI providers handle your requests and how they're prioritized. +When using [OpenRouter](https://openrouter.ai) as your LLM provider, Hermes Agent supports **provider routing** — fine-grained control over which underlying AI providers handle your requests and how they're prioritized. OpenRouter routes requests to many providers (e.g., Anthropic, Google, AWS Bedrock, Together AI). Provider routing lets you optimize for cost, speed, quality, or enforce specific provider requirements. -:::tip -Traffic routed through Nous Portal respects the same provider preferences — and Portal subscribers get 10% off token-billed providers. +:::note +[Nous Portal](/integrations/nous-portal) decides routing centrally per model and does not accept caller-supplied provider preferences; Hermes never sends the `provider` object to Portal, so `provider_routing` is simply ignored there. ::: ## Configuration @@ -30,7 +30,7 @@ provider_routing: ``` :::info -Provider routing only applies when using OpenRouter or Nous Portal. It has no effect with direct provider connections (e.g., connecting directly to the Anthropic API). +Provider routing only applies when using OpenRouter. It has no effect on Nous Portal or direct provider connections (e.g., connecting directly to the Anthropic API). ::: ## Options @@ -192,7 +192,7 @@ provider_routing: ## How It Works -Provider routing preferences are passed to OpenRouter or Nous Portal on agent chat requests and iteration-limit summaries via the `extra_body.provider` field. (`extra_body` is the OpenAI Python SDK argument; it becomes the top-level `provider` object in the JSON request.) Auxiliary tasks such as compression and title generation are configured independently under `auxiliary..extra_body`. +Provider routing preferences are passed to OpenRouter on agent chat requests and iteration-limit summaries via the `extra_body.provider` field. (`extra_body` is the OpenAI Python SDK argument; it becomes the top-level `provider` object in the JSON request.) Auxiliary tasks such as compression and title generation are configured independently under `auxiliary..extra_body`. - **CLI mode** — configured in `~/.hermes/config.yaml`, loaded at startup - **Gateway mode** — same config file, loaded when the gateway starts @@ -225,5 +225,5 @@ provider_routing: When no `provider_routing` section is configured (the default), the aggregator uses its own default routing logic, which generally balances cost and availability automatically. :::tip Provider Routing vs. Fallback Models -Provider routing controls which **sub-providers behind OpenRouter or Nous Portal** handle your requests. For automatic failover to an entirely different provider when your primary model fails, see [Fallback Providers](/user-guide/features/fallback-providers). +Provider routing controls which **sub-providers behind OpenRouter** handle your requests. For automatic failover to an entirely different provider when your primary model fails, see [Fallback Providers](/user-guide/features/fallback-providers). ::: From 81f821657c4ca40b99885bb6461b7dc2f2da666c Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 15:25:26 +0530 Subject: [PATCH 096/276] chore: map contributor email for @koltyj Attribution mapping so the cherry-picked #96187 commit resolves to its GitHub account (@jonpol01 is already in the legacy map). --- contributors/emails/krjhawks@gmail.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/krjhawks@gmail.com diff --git a/contributors/emails/krjhawks@gmail.com b/contributors/emails/krjhawks@gmail.com new file mode 100644 index 0000000000..d0fbe72d48 --- /dev/null +++ b/contributors/emails/krjhawks@gmail.com @@ -0,0 +1,2 @@ +koltyj +# PR #96187 From 63b77aa593c9ef3c3ff8fcef17357018b927fb5a Mon Sep 17 00:00:00 2001 From: Kolton Jacobs Date: Thu, 27 Aug 2026 04:04:09 -0400 Subject: [PATCH 097/276] fix(desktop): stop double-quoting expandRemotePath fragments in the SSH spawn path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit expandRemotePath() returns an already-quoted shell fragment ("$HOME"'/path'), but three call sites wrapped its output in shq() again: the withRemoteUpdateMutex python argv, the reservation/lock/ owner_file assignments in buildSpawnCommand, and the identity values in buildOwnedStaleTerminationCommand. The remote shell strips only one quoting layer, so python received a mutex path with literal quote characters in it (creating a directory literally named ' in $HOME), and the payload's mkdir "$reservation" loop spun on a path that can never exist. Every Desktop SSH backend spawn hung until the connect timeout, retried, and left an orphaned flock queue behind; stale-owner cleanup always printed REFUSED for the same reason. The regression test parses the composed command with a real sh — the same parse the remote login shell performs — and requires the mutex path and the payload's reservation paths to come out fully expanded. --- .../desktop/electron/remote-lifecycle.test.ts | 70 +++++++++++++++++++ apps/desktop/electron/remote-lifecycle.ts | 3 +- 2 files changed, 72 insertions(+), 1 deletion(-) diff --git a/apps/desktop/electron/remote-lifecycle.test.ts b/apps/desktop/electron/remote-lifecycle.test.ts index dffc716193..4ca82c2ed9 100644 --- a/apps/desktop/electron/remote-lifecycle.test.ts +++ b/apps/desktop/electron/remote-lifecycle.test.ts @@ -1849,3 +1849,73 @@ test('cleanupStale keeps the lockfile when even SIGKILL cannot confirm the pid d // The record must survive so the next connect's reap pass retries. assert.ok(!ssh.calls.some(c => /rm -f .*backend\.lock\.json/.test(c))) }) +test('buildSpawnCommand quotes expandRemotePath fragments exactly once (real sh parse)', async () => { + // Regression: expandRemotePath() returns an already-quoted shell fragment + // ("$HOME"'/path'). Wrapping such a fragment in shq() again ships literal + // quote characters to the remote: python received a mutex path named + // '"$HOME"'"'"'/…'"'"'', and the payload's `mkdir "$reservation"` spun + // forever on a path that can never exist, so every SSH backend spawn hung + // until the connect timeout. Parse the composed command with a real sh the + // way the remote login shell does, and require the paths to come out clean. + const cmd = buildSpawnCommand('/x/hermes', 'work', { + hermesHome: '~/.hermes', + logPath: spawnLogPath(OWNERSHIP_ID, SPAWN_NONCE), + ownershipId: OWNERSHIP_ID, + reservationNonce: SPAWN_NONCE, + spawnNonce: SPAWN_NONCE, + tokenFilePath: spawnTokenPath(OWNERSHIP_ID, SPAWN_NONCE), + lockMetadata: { + ownershipId: OWNERSHIP_ID, + spawnNonce: SPAWN_NONCE, + port: 0, + profile: 'work', + hermesPath: '/x/hermes', + hermesHome: '~/.hermes', + logPath: spawnLogPath(OWNERSHIP_ID, SPAWN_NONCE), + tokenFingerprint: fingerprintToken('stored-token'), + protocolVersion: PROTOCOL_VERSION, + startedAt: '2026-07-14T00:00:00.000Z' + } + }) + + // Capture the argv a remote shell would hand to python3, via a shim on PATH. + const shimDir = await mkdtemp(path.join(os.tmpdir(), 'hermes-argv-shim-')) + const fakeHome = await mkdtemp(path.join(os.tmpdir(), 'hermes-fake-home-')) + try { + const argvFile = path.join(shimDir, 'argv') + await writeFile( + path.join(shimDir, 'python3'), + `#!/bin/sh\nprintf '%s\\0' "$@" > ${argvFile}\n`, + { mode: 0o755 } + ) + await promisify(execCallback)(cmd, { + env: { ...process.env, PATH: `${shimDir}:${process.env.PATH}`, HOME: fakeHome } + }) + const argv = (await readFile(argvFile, 'utf8')).split('\0') + + // argv: ['-c', , , ] + const mutexPath = argv[2] + assert.equal( + mutexPath, + `${fakeHome}/.hermes/.hermes-update-in-progress.mutex`, + 'mutex path must reach python fully expanded, with no quote characters' + ) + + // The payload assigns reservation/lock/owner_file before its mkdir loop. + // Evaluate that prefix the way the remote sh does and require real paths. + const payload = argv[3] + const prefix = payload.slice(0, payload.indexOf('i=0;')) + const { stdout } = await promisify(execCallback)( + `${prefix} printf '%s\\n' "$reservation" "$lock" "$owner_file"`, + { env: { ...process.env, HOME: fakeHome } } + ) + const [reservation, lock, ownerFile] = stdout.split('\n') + const base = `${fakeHome}/.hermes/desktop-ssh/${OWNERSHIP_ID}` + assert.equal(reservation, `${base}/.connect.lock`) + assert.equal(lock, `${base}/backend.lock.json`) + assert.equal(ownerFile, `${base}/.connect.lock/owner`) + } finally { + await rm(shimDir, { recursive: true, force: true }) + await rm(fakeHome, { recursive: true, force: true }) + } +}) diff --git a/apps/desktop/electron/remote-lifecycle.ts b/apps/desktop/electron/remote-lifecycle.ts index d096705c95..17c98a97f5 100644 --- a/apps/desktop/electron/remote-lifecycle.ts +++ b/apps/desktop/electron/remote-lifecycle.ts @@ -913,7 +913,8 @@ finally: sys.exit(result.returncode if result is not None else 1) `.trim() - return `python3 -c ${shq(script)} ${shq(mutexPath)} ${shq(command)}` + // mutexPath is an expandRemotePath fragment — already quoted; do not shq again. + return `python3 -c ${shq(script)} ${mutexPath} ${shq(command)}` } /** From b98edf8898e0eb8454a317410caf99c71e6f799d Mon Sep 17 00:00:00 2001 From: John Paul Soliva Date: Thu, 3 Sep 2026 22:12:40 +0900 Subject: [PATCH 098/276] chore(desktop): drop the committed update-mutex artifact directory `apps/desktop/'` is not a real path. It is a literal single-quote directory holding a full absolute path as nested subdirectories: apps/desktop/'/var/folders/5h/.../hermes-update-mutex-LMF9y5/home/.hermes-update-in-progress.mutex' Both files are 0 bytes, nothing in the tree references them, and the leading and trailing `'` are part of the filenames. They are the fingerprint of the double-quoted mutex path in `withRemoteUpdateMutex()`: the path reaches Python with its shell quotes still attached, so it is treated as CWD-relative and `os.makedirs()` materialises the whole absolute path under whatever directory the process happened to be in. Running `apps/desktop/electron/remote-lifecycle.test.ts` from `apps/desktop` reproduces it on the spot. They were swept in by a `git add .` in 36620578f0, an unrelated menu-label commit. This only removes the committed artifact. The generator is a separate concern already covered by open PRs (#99189, #96187, #96260) and issues (#99133, #96212, #96188); those repair the quoting but none of them deletes these two files, so the litter would survive whichever one lands. --- .../home/.hermes-update-in-progress.mutex' | 0 .../home/.hermes-update-in-progress.mutex' | 0 2 files changed, 0 insertions(+), 0 deletions(-) delete mode 100644 apps/desktop/'/var/folders/5h/qzgt02rn619fttp2d7zdxj600000gn/T/hermes-update-mutex-LMF9y5/home/.hermes-update-in-progress.mutex' delete mode 100644 apps/desktop/'/var/folders/5h/qzgt02rn619fttp2d7zdxj600000gn/T/hermes-update-mutex-us8HZu/home/.hermes-update-in-progress.mutex' diff --git a/apps/desktop/'/var/folders/5h/qzgt02rn619fttp2d7zdxj600000gn/T/hermes-update-mutex-LMF9y5/home/.hermes-update-in-progress.mutex' b/apps/desktop/'/var/folders/5h/qzgt02rn619fttp2d7zdxj600000gn/T/hermes-update-mutex-LMF9y5/home/.hermes-update-in-progress.mutex' deleted file mode 100644 index e69de29bb2..0000000000 diff --git a/apps/desktop/'/var/folders/5h/qzgt02rn619fttp2d7zdxj600000gn/T/hermes-update-mutex-us8HZu/home/.hermes-update-in-progress.mutex' b/apps/desktop/'/var/folders/5h/qzgt02rn619fttp2d7zdxj600000gn/T/hermes-update-mutex-us8HZu/home/.hermes-update-in-progress.mutex' deleted file mode 100644 index e69de29bb2..0000000000 From 06402ecb7ca5c583f035941a88acc803ddaf9118 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 15:34:59 +0530 Subject: [PATCH 099/276] test(desktop): tighten the real-sh mutex quoting test Review follow-ups on the #96187 cherry-pick: skip on win32 like the sibling tests that shell out; reuse the file's `exec` helper; one temp root so a failing second mkdtemp cannot leak the shim dir; quote the shim's redirect target; guard the payload-prefix sentinel so a renamed loop marker can never make the test execute the real spawn payload; drop the lockMetadata fields the assertions never read. Move the mutexPath contract into withRemoteUpdateMutex's doc comment instead of a third inline restatement. --- .../desktop/electron/remote-lifecycle.test.ts | 58 +++++++------------ apps/desktop/electron/remote-lifecycle.ts | 5 +- 2 files changed, 23 insertions(+), 40 deletions(-) diff --git a/apps/desktop/electron/remote-lifecycle.test.ts b/apps/desktop/electron/remote-lifecycle.test.ts index 4ca82c2ed9..6cd596ae85 100644 --- a/apps/desktop/electron/remote-lifecycle.test.ts +++ b/apps/desktop/electron/remote-lifecycle.test.ts @@ -1849,14 +1849,10 @@ test('cleanupStale keeps the lockfile when even SIGKILL cannot confirm the pid d // The record must survive so the next connect's reap pass retries. assert.ok(!ssh.calls.some(c => /rm -f .*backend\.lock\.json/.test(c))) }) -test('buildSpawnCommand quotes expandRemotePath fragments exactly once (real sh parse)', async () => { - // Regression: expandRemotePath() returns an already-quoted shell fragment - // ("$HOME"'/path'). Wrapping such a fragment in shq() again ships literal - // quote characters to the remote: python received a mutex path named - // '"$HOME"'"'"'/…'"'"'', and the payload's `mkdir "$reservation"` spun - // forever on a path that can never exist, so every SSH backend spawn hung - // until the connect timeout. Parse the composed command with a real sh the - // way the remote login shell does, and require the paths to come out clean. +test.skipIf(process.platform === 'win32')('buildSpawnCommand quotes expandRemotePath fragments exactly once (real sh parse)', async () => { + // expandRemotePath() output is pre-quoted; a second shq() ships literal quote + // characters to the remote python. Parse the composed command with a real sh, + // as the remote login shell does, and require every path to come out clean. const cmd = buildSpawnCommand('/x/hermes', 'work', { hermesHome: '~/.hermes', logPath: spawnLogPath(OWNERSHIP_ID, SPAWN_NONCE), @@ -1864,58 +1860,44 @@ test('buildSpawnCommand quotes expandRemotePath fragments exactly once (real sh reservationNonce: SPAWN_NONCE, spawnNonce: SPAWN_NONCE, tokenFilePath: spawnTokenPath(OWNERSHIP_ID, SPAWN_NONCE), - lockMetadata: { - ownershipId: OWNERSHIP_ID, - spawnNonce: SPAWN_NONCE, - port: 0, - profile: 'work', - hermesPath: '/x/hermes', - hermesHome: '~/.hermes', - logPath: spawnLogPath(OWNERSHIP_ID, SPAWN_NONCE), - tokenFingerprint: fingerprintToken('stored-token'), - protocolVersion: PROTOCOL_VERSION, - startedAt: '2026-07-14T00:00:00.000Z' - } + lockMetadata: { ownershipId: OWNERSHIP_ID, spawnNonce: SPAWN_NONCE } }) // Capture the argv a remote shell would hand to python3, via a shim on PATH. - const shimDir = await mkdtemp(path.join(os.tmpdir(), 'hermes-argv-shim-')) - const fakeHome = await mkdtemp(path.join(os.tmpdir(), 'hermes-fake-home-')) + const root = await mkdtemp(path.join(os.tmpdir(), 'hermes-argv-shim-')) try { + const shimDir = path.join(root, 'shim') + const fakeHome = path.join(root, 'home') + await mkdir(shimDir) + await mkdir(fakeHome) const argvFile = path.join(shimDir, 'argv') - await writeFile( - path.join(shimDir, 'python3'), - `#!/bin/sh\nprintf '%s\\0' "$@" > ${argvFile}\n`, - { mode: 0o755 } - ) - await promisify(execCallback)(cmd, { + await writeFile(path.join(shimDir, 'python3'), `#!/bin/sh\nprintf '%s\\0' "$@" > '${argvFile}'\n`, { mode: 0o755 }) + await exec(cmd, { env: { ...process.env, PATH: `${shimDir}:${process.env.PATH}`, HOME: fakeHome } }) const argv = (await readFile(argvFile, 'utf8')).split('\0') // argv: ['-c', , , ] - const mutexPath = argv[2] assert.equal( - mutexPath, + argv[2], `${fakeHome}/.hermes/.hermes-update-in-progress.mutex`, 'mutex path must reach python fully expanded, with no quote characters' ) // The payload assigns reservation/lock/owner_file before its mkdir loop. - // Evaluate that prefix the way the remote sh does and require real paths. + // Evaluate only that prefix the way the remote sh does; never the loop itself. const payload = argv[3] - const prefix = payload.slice(0, payload.indexOf('i=0;')) - const { stdout } = await promisify(execCallback)( - `${prefix} printf '%s\\n' "$reservation" "$lock" "$owner_file"`, - { env: { ...process.env, HOME: fakeHome } } - ) + const loopStart = payload.indexOf('i=0;') + assert.ok(loopStart > 0, 'payload prefix sentinel missing') + const { stdout } = await exec(`${payload.slice(0, loopStart)} printf '%s\\n' "$reservation" "$lock" "$owner_file"`, { + env: { ...process.env, HOME: fakeHome } + }) const [reservation, lock, ownerFile] = stdout.split('\n') const base = `${fakeHome}/.hermes/desktop-ssh/${OWNERSHIP_ID}` assert.equal(reservation, `${base}/.connect.lock`) assert.equal(lock, `${base}/backend.lock.json`) assert.equal(ownerFile, `${base}/.connect.lock/owner`) } finally { - await rm(shimDir, { recursive: true, force: true }) - await rm(fakeHome, { recursive: true, force: true }) + await rm(root, { recursive: true, force: true }) } }) diff --git a/apps/desktop/electron/remote-lifecycle.ts b/apps/desktop/electron/remote-lifecycle.ts index 17c98a97f5..40108383cd 100644 --- a/apps/desktop/electron/remote-lifecycle.ts +++ b/apps/desktop/electron/remote-lifecycle.ts @@ -895,7 +895,9 @@ finally: // the marker check, spawns the backend, and publishes its initial lockfile. // Python keeps the descriptor close-on-exec by default and passes it explicitly // only to the intended outer shell; each detached child closes it before -// execing Hermes. +// execing Hermes. mutexPath is expandRemotePath() output — a complete shell +// word ("$HOME"'/…' or '/abs/…') embedded raw so $HOME expands remotely; a +// second shq() would hand python the quote characters as part of the path. function withRemoteUpdateMutex(command, mutexPath) { const script = ` import fcntl,os,subprocess,sys @@ -913,7 +915,6 @@ finally: sys.exit(result.returncode if result is not None else 1) `.trim() - // mutexPath is an expandRemotePath fragment — already quoted; do not shq again. return `python3 -c ${shq(script)} ${mutexPath} ${shq(command)}` } From 089bb32886c8c18f7fa20182c7bf8826d6935ac5 Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Sun, 6 Sep 2026 10:15:23 +0000 Subject: [PATCH 100/276] fmt(js): `npm run fix` on merge (#104210) Co-authored-by: github-actions[bot] --- .../desktop/electron/remote-lifecycle.test.ts | 107 ++++++++++-------- .../desktop/src/store/statusbar-prefs.test.ts | 5 +- 2 files changed, 63 insertions(+), 49 deletions(-) diff --git a/apps/desktop/electron/remote-lifecycle.test.ts b/apps/desktop/electron/remote-lifecycle.test.ts index 6cd596ae85..408cd9e27a 100644 --- a/apps/desktop/electron/remote-lifecycle.test.ts +++ b/apps/desktop/electron/remote-lifecycle.test.ts @@ -1849,55 +1849,66 @@ test('cleanupStale keeps the lockfile when even SIGKILL cannot confirm the pid d // The record must survive so the next connect's reap pass retries. assert.ok(!ssh.calls.some(c => /rm -f .*backend\.lock\.json/.test(c))) }) -test.skipIf(process.platform === 'win32')('buildSpawnCommand quotes expandRemotePath fragments exactly once (real sh parse)', async () => { - // expandRemotePath() output is pre-quoted; a second shq() ships literal quote - // characters to the remote python. Parse the composed command with a real sh, - // as the remote login shell does, and require every path to come out clean. - const cmd = buildSpawnCommand('/x/hermes', 'work', { - hermesHome: '~/.hermes', - logPath: spawnLogPath(OWNERSHIP_ID, SPAWN_NONCE), - ownershipId: OWNERSHIP_ID, - reservationNonce: SPAWN_NONCE, - spawnNonce: SPAWN_NONCE, - tokenFilePath: spawnTokenPath(OWNERSHIP_ID, SPAWN_NONCE), - lockMetadata: { ownershipId: OWNERSHIP_ID, spawnNonce: SPAWN_NONCE } - }) - - // Capture the argv a remote shell would hand to python3, via a shim on PATH. - const root = await mkdtemp(path.join(os.tmpdir(), 'hermes-argv-shim-')) - try { - const shimDir = path.join(root, 'shim') - const fakeHome = path.join(root, 'home') - await mkdir(shimDir) - await mkdir(fakeHome) - const argvFile = path.join(shimDir, 'argv') - await writeFile(path.join(shimDir, 'python3'), `#!/bin/sh\nprintf '%s\\0' "$@" > '${argvFile}'\n`, { mode: 0o755 }) - await exec(cmd, { - env: { ...process.env, PATH: `${shimDir}:${process.env.PATH}`, HOME: fakeHome } +test.skipIf(process.platform === 'win32')( + 'buildSpawnCommand quotes expandRemotePath fragments exactly once (real sh parse)', + async () => { + // expandRemotePath() output is pre-quoted; a second shq() ships literal quote + // characters to the remote python. Parse the composed command with a real sh, + // as the remote login shell does, and require every path to come out clean. + const cmd = buildSpawnCommand('/x/hermes', 'work', { + hermesHome: '~/.hermes', + logPath: spawnLogPath(OWNERSHIP_ID, SPAWN_NONCE), + ownershipId: OWNERSHIP_ID, + reservationNonce: SPAWN_NONCE, + spawnNonce: SPAWN_NONCE, + tokenFilePath: spawnTokenPath(OWNERSHIP_ID, SPAWN_NONCE), + lockMetadata: { ownershipId: OWNERSHIP_ID, spawnNonce: SPAWN_NONCE } }) - const argv = (await readFile(argvFile, 'utf8')).split('\0') - // argv: ['-c', , , ] - assert.equal( - argv[2], - `${fakeHome}/.hermes/.hermes-update-in-progress.mutex`, - 'mutex path must reach python fully expanded, with no quote characters' - ) + // Capture the argv a remote shell would hand to python3, via a shim on PATH. + const root = await mkdtemp(path.join(os.tmpdir(), 'hermes-argv-shim-')) - // The payload assigns reservation/lock/owner_file before its mkdir loop. - // Evaluate only that prefix the way the remote sh does; never the loop itself. - const payload = argv[3] - const loopStart = payload.indexOf('i=0;') - assert.ok(loopStart > 0, 'payload prefix sentinel missing') - const { stdout } = await exec(`${payload.slice(0, loopStart)} printf '%s\\n' "$reservation" "$lock" "$owner_file"`, { - env: { ...process.env, HOME: fakeHome } - }) - const [reservation, lock, ownerFile] = stdout.split('\n') - const base = `${fakeHome}/.hermes/desktop-ssh/${OWNERSHIP_ID}` - assert.equal(reservation, `${base}/.connect.lock`) - assert.equal(lock, `${base}/backend.lock.json`) - assert.equal(ownerFile, `${base}/.connect.lock/owner`) - } finally { - await rm(root, { recursive: true, force: true }) + try { + const shimDir = path.join(root, 'shim') + const fakeHome = path.join(root, 'home') + await mkdir(shimDir) + await mkdir(fakeHome) + const argvFile = path.join(shimDir, 'argv') + await writeFile(path.join(shimDir, 'python3'), `#!/bin/sh\nprintf '%s\\0' "$@" > '${argvFile}'\n`, { + mode: 0o755 + }) + await exec(cmd, { + env: { ...process.env, PATH: `${shimDir}:${process.env.PATH}`, HOME: fakeHome } + }) + const argv = (await readFile(argvFile, 'utf8')).split('\0') + + // argv: ['-c', , , ] + assert.equal( + argv[2], + `${fakeHome}/.hermes/.hermes-update-in-progress.mutex`, + 'mutex path must reach python fully expanded, with no quote characters' + ) + + // The payload assigns reservation/lock/owner_file before its mkdir loop. + // Evaluate only that prefix the way the remote sh does; never the loop itself. + const payload = argv[3] + const loopStart = payload.indexOf('i=0;') + assert.ok(loopStart > 0, 'payload prefix sentinel missing') + + const { stdout } = await exec( + `${payload.slice(0, loopStart)} printf '%s\\n' "$reservation" "$lock" "$owner_file"`, + { + env: { ...process.env, HOME: fakeHome } + } + ) + + const [reservation, lock, ownerFile] = stdout.split('\n') + const base = `${fakeHome}/.hermes/desktop-ssh/${OWNERSHIP_ID}` + assert.equal(reservation, `${base}/.connect.lock`) + assert.equal(lock, `${base}/backend.lock.json`) + assert.equal(ownerFile, `${base}/.connect.lock/owner`) + } finally { + await rm(root, { recursive: true, force: true }) + } } -}) +) diff --git a/apps/desktop/src/store/statusbar-prefs.test.ts b/apps/desktop/src/store/statusbar-prefs.test.ts index a8f67417d9..afe111e4f0 100644 --- a/apps/desktop/src/store/statusbar-prefs.test.ts +++ b/apps/desktop/src/store/statusbar-prefs.test.ts @@ -38,7 +38,10 @@ describe('statusbar hidden items', () => { }) it('surfaces the approval pill for installs that hid it under the v1 defaults, keeping their other choices', async () => { - window.localStorage.setItem('hermes.desktop.statusbarHidden', JSON.stringify(['approval-mode', 'cron', 'gateway-health'])) + window.localStorage.setItem( + 'hermes.desktop.statusbarHidden', + JSON.stringify(['approval-mode', 'cron', 'gateway-health']) + ) const { $statusbarHiddenIds } = await loadStore() From 0cb996d977187b7e82e2d7126a0018dc6d9d5ae9 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:39:31 -0700 Subject: [PATCH 101/276] fix(delegation): number delegation batches per conversation, not per process MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `format_batch_tag` handed out `set N` ordinals from one process-wide table, so every conversation on a shared backend and every child's nested fan-out advanced the same counter. A user's second wave of 15 lanes rendered as `[set 20 · 13/15]`, which reads like 20 batches were spawned. Scope the ordinal table by the parent conversation (`parent_agent.session_id`) and thread the parent through the three render sites (batch header, completion lines, child tree-line prefix via the shared session_ref). The first fan-out in a conversation is `set 1`, the next `set 2`; sibling conversations and nested child batches no longer inflate it. Invariant test proven red on origin/main, green with the fix. --- tests/tools/test_delegate_batch_tag.py | 14 +++++++++++ tools/delegate_tool_dispatch.py | 4 ++-- tools/delegate_tool_progress.py | 33 +++++++++++++++++--------- 3 files changed, 38 insertions(+), 13 deletions(-) diff --git a/tests/tools/test_delegate_batch_tag.py b/tests/tools/test_delegate_batch_tag.py index 05d1e43593..1b31f16969 100644 --- a/tests/tools/test_delegate_batch_tag.py +++ b/tests/tools/test_delegate_batch_tag.py @@ -28,6 +28,20 @@ def test_format_batch_tag_assigns_stable_ordinals_per_batch(): assert format_batch_tag("") == "" +def test_batch_ordinals_are_scoped_per_parent_conversation(): + """One process hosts many conversations plus every child's nested fan-out; a user's + second wave must read ``set 2``, not the process-wide count of all batches ever seen.""" + parent_a = types.SimpleNamespace(session_id="conv-a") + parent_b = types.SimpleNamespace(session_id="conv-b") + assert format_batch_tag("deleg_a1", parent_a) == "set 1" + # Sibling conversation and a child's own fan-out interleave on the same process... + assert format_batch_tag("deleg_b1", parent_b) == "set 1" + assert format_batch_tag("deleg_b2", parent_b) == "set 2" + # ...without inflating parent A's next wave. + assert format_batch_tag("deleg_a2", parent_a) == "set 2" + assert format_batch_tag("deleg_a1", parent_a) == "set 1" # stable + + @pytest.mark.parametrize( "deleg, idx, count, expected", [ diff --git a/tools/delegate_tool_dispatch.py b/tools/delegate_tool_dispatch.py index 5f883a2b8f..03093da0e6 100644 --- a/tools/delegate_tool_dispatch.py +++ b/tools/delegate_tool_dispatch.py @@ -58,7 +58,7 @@ class _Batch: def _announce_batch(parent_agent, n_tasks: int, live_deleg_id: Optional[str]) -> None: """Announce the batch tag once so interleaved ``[tag n/N]`` lines are attributable.""" if n_tasks > 1 and live_deleg_id: - _hdr = f" 🔀 [{format_batch_tag(live_deleg_id)}] delegating {n_tasks} tasks" + _hdr = f" 🔀 [{format_batch_tag(live_deleg_id, parent_agent)}] delegating {n_tasks} tasks" _print_completion_line(parent_agent, getattr(parent_agent, "_delegate_spinner", None), _hdr, console_line=_hdr) def _capture_origin() -> tuple[str, str, Any, Any]: @@ -100,7 +100,7 @@ def _run_children_parallel(batch: _Batch, results: list, *, honor_parent_interru parent_agent, n_tasks = batch.parent_agent, len(batch.task_list) task_labels = [t["goal"][:40] for t in batch.task_list] spinner_ref = getattr(parent_agent, "_delegate_spinner", None) - _tag = format_batch_tag(batch.live_deleg_id) + _tag = format_batch_tag(batch.live_deleg_id, parent_agent) # Fabricated entries for still-pending / raised futures carry the correct _delegate_role. _child_by_index = {i: child for (i, _, child) in batch.children} diff --git a/tools/delegate_tool_progress.py b/tools/delegate_tool_progress.py index f76f572522..f9823c11cc 100644 --- a/tools/delegate_tool_progress.py +++ b/tools/delegate_tool_progress.py @@ -198,25 +198,29 @@ def _resolve_workspace_hint(parent_agent) -> Optional[str]: return text return None -_BATCH_ORDINALS: Dict[str, int] = {} +_BATCH_ORDINALS: Dict[str, Dict[str, int]] = {} _BATCH_ORDINALS_LOCK = threading.Lock() -def format_batch_tag(delegation_id: Optional[str]) -> str: - """Short human tag for a delegation batch: ``deleg_6a664903`` → ``set 1`` (first batch seen in this process), the - next distinct id → ``set 2``. Several batches (a parent's fan-out plus a child's nested fan-out, or two +def format_batch_tag(delegation_id: Optional[str], parent_agent: Any = None) -> str: + """Short human tag for a delegation batch: the parent's first fan-out is ``set 1``, its next distinct + delegation id ``set 2``. Several batches (a parent's fan-out plus a child's nested fan-out, or two concurrent tools) print interleaved ``[n/N]`` lines to one console; without a tag ``✓ [3/3]`` and ``✓ [3/9]`` - are indistinguishable, and a raw hex slice is unreadable. Empty string when no id is known so callers can - concatenate unconditionally.""" + are indistinguishable, and a raw hex slice is unreadable. Ordinals are scoped per parent conversation + (``parent_agent.session_id``): one process hosts many conversations and every child's own fan-out, so a + process-wide counter showed a user's second wave as ``set 20``. Empty string when no id is known so callers + can concatenate unconditionally.""" if not isinstance(delegation_id, str) or not delegation_id: return "" + scope = str(getattr(parent_agent, "session_id", None) or "") with _BATCH_ORDINALS_LOCK: - n = _BATCH_ORDINALS.setdefault(delegation_id, len(_BATCH_ORDINALS) + 1) + ordinals = _BATCH_ORDINALS.setdefault(scope, {}) + n = ordinals.setdefault(delegation_id, len(ordinals) + 1) return f"set {n}" -def _batch_prefix(delegation_id: Optional[str], task_index: int, task_count: int) -> str: +def _batch_prefix(delegation_id: Optional[str], task_index: int, task_count: int, parent_agent: Any = None) -> str: """``[set 2 · 3/9] `` for batch children, ``[set 2] `` for a lone child, ``[3/9] `` / ``""`` when the batch id is unknown.""" - tag = format_batch_tag(delegation_id) + tag = format_batch_tag(delegation_id, parent_agent) if task_count > 1: inner = f"{tag} · {task_index + 1}/{task_count}" if tag else f"{task_index + 1}/{task_count}" return f"[{inner}] " @@ -268,8 +272,12 @@ class _ChildProgressRelay: def _prefix(self) -> str: # The batch tag is resolved lazily from session_ref: the relay is built - # before delegate_task stamps ``_delegation_id`` on the child. - return _batch_prefix(self.session_ref.get("delegation_id"), self.task_index, self.task_count) + # before delegate_task stamps ``_delegation_id`` on the child. The parent + # scope rides in the same ref so ordinals match the parent's own lines. + return _batch_prefix( + self.session_ref.get("delegation_id"), self.task_index, self.task_count, + parent_agent=self.session_ref.get("_parent_scope"), + ) def _identity_kwargs(self) -> Dict[str, Any]: kw: Dict[str, Any] = {"task_index": self.task_index, "task_count": self.task_count, "goal": self.goal_label} @@ -375,6 +383,9 @@ def _build_child_progress_callback( parent_cb = getattr(parent_agent, "tool_progress_callback", None) if not spinner and not parent_cb: return None + if session_ref is not None: + # Not an identity kwarg (underscore-prefixed, never relayed); only scopes the batch ordinal. + session_ref["_parent_scope"] = parent_agent return _ChildProgressRelay( task_index, goal, spinner, parent_cb, task_count, subagent_id, parent_id, depth, model, toolsets, session_ref, ) From 3009efe75e772646858ee7ba42c4d21dae02f518 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:18:06 -0700 Subject: [PATCH 102/276] fix(tools): read_file dispatch honors the advertised 2000-line default when the model omits limit #76996 raised DEFAULT_READ_LIMIT, the read_file_tool signature and the schema default from 500 to 2000, but the registry dispatch handler `_handle_read_file` kept its own literal `args.get("limit", 500)`. Since models omit `limit` on most reads, every dispatched read still stopped at 500 lines with `truncated: true`, so the default flip never reached production traffic. All three sites now read the single DEFAULT_READ_LIMIT constant, so the schema, the Python default and the dispatch fallback cannot drift again. Live probe (registry.dispatch on a 1500-line file, no limit): before 501 lines / truncated=true; after 1501 lines / truncated=false. --- tests/tools/test_file_tools.py | 16 ++++++++++++++++ tools/file_tools.py | 7 ++++--- 2 files changed, 20 insertions(+), 3 deletions(-) diff --git a/tests/tools/test_file_tools.py b/tests/tools/test_file_tools.py index 762b6d7fda..48f22f381d 100644 --- a/tests/tools/test_file_tools.py +++ b/tests/tools/test_file_tools.py @@ -32,6 +32,22 @@ class TestReadFileHandler: mock_ops.read_file.assert_called_once_with("/tmp/test.txt", 1, 2000) + @patch("tools.file_tools._get_file_ops") + def test_dispatch_without_limit_uses_schema_default(self, mock_get): + """The model omits ``limit`` on most reads; the dispatch handler must fall + back to the SAME default the schema advertises (drifted to 500 vs 2000).""" + from tools.file_tools import READ_FILE_SCHEMA, _handle_read_file + mock_ops = MagicMock() + result_obj = MagicMock() + result_obj.content = "x" + result_obj.to_dict.return_value = {"content": "x", "total_lines": 1} + mock_ops.read_file.return_value = result_obj + mock_get.return_value = mock_ops + + _handle_read_file({"path": "/tmp/test.txt"}, task_id="t-default") + schema_default = READ_FILE_SCHEMA["parameters"]["properties"]["limit"]["default"] + assert mock_ops.read_file.call_args.args[2] == schema_default + @patch("tools.file_tools._get_file_ops") def test_exception_returns_error_json(self, mock_get): mock_get.side_effect = RuntimeError("terminal not available") diff --git a/tools/file_tools.py b/tools/file_tools.py index e355a6eede..38d210b8b8 100644 --- a/tools/file_tools.py +++ b/tools/file_tools.py @@ -22,6 +22,7 @@ from agent.file_safety import get_read_block_error from tools.binary_extensions import has_binary_extension from tools.file_operations import ( ShellFileOperations, normalize_read_pagination, normalize_search_pagination) +from tools.file_operations_common import DEFAULT_READ_LIMIT from tools import file_state from agent.redact import redact_sensitive_text from tools.file_tools_paths import ( @@ -536,7 +537,7 @@ def _record_successful_read(task_data: dict, task_id: str, path: str, resolved_s return count -def read_file_tool(path: str, offset: int = 1, limit: int = 2000, task_id: str = "default") -> str: +def read_file_tool(path: str, offset: int = 1, limit: int = DEFAULT_READ_LIMIT, task_id: str = "default") -> str: """Read a file with pagination and line numbers. Guard order: device-path blocklist (no I/O) → stat-based special-file @@ -979,7 +980,7 @@ READ_FILE_SCHEMA = { "properties": { "path": {"type": "string", "description": "Path to the file to read (absolute, relative, or ~/path)"}, "offset": {"type": "integer", "description": "Line number to start reading from (1-indexed, default: 1)", "default": 1, "minimum": 1}, - "limit": {"type": "integer", "description": "Maximum number of lines to read (default: 2000, max: 2000). Reads are additionally capped at a ~100K-character budget with a next_offset continuation.", "default": 2000, "maximum": 2000} + "limit": {"type": "integer", "description": "Maximum number of lines to read (default: 2000, max: 2000). Reads are additionally capped at a ~100K-character budget with a next_offset continuation.", "default": DEFAULT_READ_LIMIT, "maximum": 2000} }, "required": ["path"] } @@ -1123,7 +1124,7 @@ SEARCH_FILES_SCHEMA = { def _handle_read_file(args, **kw): tid = kw.get("task_id") or "default" - return read_file_tool(path=args.get("path", ""), offset=args.get("offset", 1), limit=args.get("limit", 500), task_id=tid) + return read_file_tool(path=args.get("path", ""), offset=args.get("offset", 1), limit=args.get("limit", DEFAULT_READ_LIMIT), task_id=tid) def _handle_write_file(args, **kw): From e8521f47f97ad4782706d2612b54ad7227c17a1b Mon Sep 17 00:00:00 2001 From: schrodienieur Date: Sat, 5 Sep 2026 00:18:22 +0700 Subject: [PATCH 103/276] fix(mem0): lazy-import _read_mem0_json to avoid ImportError during plugin load The plugin loader pre-registers empty shells in sys.modules before executing any submodule. When _setup.py ran 'from . import _read_mem0_json' at module level, Python found the empty parent shell (init not yet executed), so the import failed silently. This left _setup.py partially loaded and post_setup undefined, causing 'cannot import name post_setup' during hermes memory setup mem0. Move the import inside the three functions that use it, matching the late-import pattern __init__.py already uses for post_setup. Fixes: hermes memory setup mem0 crashing with ImportError --- plugins/memory/mem0/_setup.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/plugins/memory/mem0/_setup.py b/plugins/memory/mem0/_setup.py index 1f0f25d261..b623baa9c8 100644 --- a/plugins/memory/mem0/_setup.py +++ b/plugins/memory/mem0/_setup.py @@ -18,7 +18,6 @@ from typing import Any from hermes_constants import get_hermes_home # noqa: F401 — patched by tests -from . import _read_mem0_json from ._oss_providers import EMBEDDER_PROVIDERS, KNOWN_DIMS, LLM_PROVIDERS, SECTION_REGISTRIES, VECTOR_PROVIDERS, validate_oss_config _OLLAMA_URL = "http://localhost:11434" @@ -170,6 +169,7 @@ def _persist_provider_config(hermes_home: str, config: dict, provider_config: di def _setup_platform(hermes_home: str, config: dict, flags: dict[str, str]) -> None: """Platform mode setup — prompts for API key (secret -> .env), user/agent ids and rerank (-> mem0.json).""" + from . import _read_mem0_json provider_config = _read_mem0_json(Path(hermes_home) / "mem0.json") print("\n Configuring mem0:\n") env_writes = _api_key_writes(flags, "Mem0 Platform API key", url="https://app.mem0.ai") @@ -205,6 +205,7 @@ def _check_selfhosted_server(host: str) -> None: def _setup_selfhosted(hermes_home: str, config: dict, flags: dict[str, str]) -> None: """Self-hosted mode — point at an existing Mem0 server: URL -> mem0.json, key -> .env (MEM0_API_KEY).""" + from . import _read_mem0_json provider_config = _read_mem0_json(Path(hermes_home) / "mem0.json") print("\n Configuring mem0 (self-hosted server):\n") host = flags.get("host") or _prompt("Mem0 server URL (e.g. http://localhost:8888)", default=provider_config.get("host") or None) @@ -237,6 +238,7 @@ def _print_oss_summary(oss_config: dict, env_writes: dict, dry_run: bool = False def _finish_oss(hermes_home: str, config: dict, oss_config: dict, env_writes: dict[str, str], user_id: str, agent_id: str, pgvector_config: dict | None = None) -> None: """Shared OSS tail: write secrets + mem0.json, install deps, activate, check, summarize.""" + from . import _read_mem0_json if env_writes: _write_env(Path(hermes_home) / ".env", env_writes) config_path = Path(hermes_home) / "mem0.json" # merge-write, plain text (platform path uses save_config's 0600 atomic write) From c8cbc07030459162a85058a57f8155c4d9efcde0 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:21:43 -0700 Subject: [PATCH 104/276] test(mem0): setup module keeps post_setup when discovery imports the package first MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The plugin loader execs sibling modules before the package __init__, so a module-level `from . import _read_mem0_json` in _setup.py failed against the empty parent shell and the whole module silently dropped out — every later `hermes memory setup mem0` died with "cannot import name 'post_setup'". The test loads mem0 through the real discovery path from a cold sys.modules and asserts the cached _setup module exposes post_setup (red on base). Also maps the contributor email for #103078 credit. Campaign tracker: https://github.com/NousResearch/hermes-agent/issues/104154 --- .../emails/muhammadnizamuddinaulia@gmail.com | 2 ++ tests/plugins/memory/test_mem0_setup.py | 18 ++++++++++++++++++ 2 files changed, 20 insertions(+) create mode 100644 contributors/emails/muhammadnizamuddinaulia@gmail.com diff --git a/contributors/emails/muhammadnizamuddinaulia@gmail.com b/contributors/emails/muhammadnizamuddinaulia@gmail.com new file mode 100644 index 0000000000..98157426ab --- /dev/null +++ b/contributors/emails/muhammadnizamuddinaulia@gmail.com @@ -0,0 +1,2 @@ +schrodienieur +# PR #103078 salvage diff --git a/tests/plugins/memory/test_mem0_setup.py b/tests/plugins/memory/test_mem0_setup.py index 8242fec2fc..d265c3bbbe 100644 --- a/tests/plugins/memory/test_mem0_setup.py +++ b/tests/plugins/memory/test_mem0_setup.py @@ -253,3 +253,21 @@ class TestConnectivityChecks: assert ok is True + + +def test_discovery_loaded_setup_module_exposes_post_setup(monkeypatch): + """`hermes memory setup mem0` reaches the wizard when the package is first imported by plugin + discovery, which execs sibling modules before ``__init__`` (#103078). The invariant is on the + module the loader actually cached, not on a normal top-level import.""" + from plugins.memory import load_memory_provider + + saved = {k: sys.modules.pop(k) for k in list(sys.modules) if k.startswith("plugins.memory.mem0")} + try: + provider = load_memory_provider("mem0", register_skills=False) + assert provider is not None + assert hasattr(sys.modules["plugins.memory.mem0._setup"], "post_setup") + finally: + for k in list(sys.modules): + if k.startswith("plugins.memory.mem0"): + del sys.modules[k] + sys.modules.update(saved) From 1f16bbf108a84c4cf6a512383edf6f2718b871d4 Mon Sep 17 00:00:00 2001 From: Edizzier Date: Sun, 6 Sep 2026 09:11:02 +0300 Subject: [PATCH 105/276] fix(cli): refuse to persist a truncated JSON reply as profile description hermes_cli/profile_describer.py's aux-LLM response parser is documented as "lenient, never raises": when the reply doesn't parse as JSON via _extract_json_blob, it falls back to treating the WHOLE raw reply as plain prose and persists it (truncated to 280 chars) as profile.yaml's description. That fallback exists for models that ignore the JSON-output instruction and just answer in prose -- reasonable. But it doesn't distinguish that case from a reply that DID start out as the requested JSON object and was cut off mid-object by the aux model or transport. _extract_json_blob requires a matching closing brace, so a truncated object (no `}` at all, or one cut off mid-string) returns None just like plain prose does, and the fallback then persists the raw JSON fragment verbatim: literal leading `{`, `\n` escapes, a sentence chopped mid-word (#104067). Fix: when parsing fails, check whether the reply (after the existing code-fence strip) still looks JSON-shaped (starts with `{`). If so, it's a malformed/truncated structured reply, not prose -- refuse it and return ok=False instead of persisting the fragment. Only fall through to the raw-text-as-prose fallback when the reply never looked like JSON to begin with, so genuinely prose-only aux replies keep working exactly as before. Also logs one INFO line on the refusal path, per the issue's note that the silent fallback made this undiagnosable. Update (review feedback from @jerrygooch): profile_describer._FENCE_RE was case-sensitive, so an uppercase ```JSON fence wasn't stripped -- stripping left "JSON\n{..." which doesn't start with "{", so a truncated response under that fence variant still fell through to the prose fallback and got persisted, defeating the fix for that case. Added re.IGNORECASE to _FENCE_RE, aligning it with kanban_specify._FENCE_RE (which already used re.IGNORECASE), and added a regression test for the uppercase-fence case. Added tests/hermes_cli/test_profile_describer.py coverage: a truncated object cut off mid-sentence (the real-world shape from the issue), one cut off right after the opening brace, the same truncation under a ```json fence, the same truncation under an uppercase ```JSON fence, and a regression guard that a genuine plain-prose reply still hits the existing lenient fallback unchanged. Fixes #104067 Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01LUtGTop5iGKi5vfWnKnwET --- hermes_cli/profile_describer.py | 15 +++- tests/hermes_cli/test_profile_describer.py | 97 ++++++++++++++++++++++ 2 files changed, 111 insertions(+), 1 deletion(-) diff --git a/hermes_cli/profile_describer.py b/hermes_cli/profile_describer.py index 97f04aa9f9..2e17e0b19f 100644 --- a/hermes_cli/profile_describer.py +++ b/hermes_cli/profile_describer.py @@ -62,7 +62,7 @@ Notable skills (up to {skill_cap}): """ -_FENCE_RE = re.compile(r"^```(?:json)?\s*|\s*```$", re.MULTILINE) +_FENCE_RE = re.compile(r"^```(?:json)?\s*|\s*```$", re.MULTILINE | re.IGNORECASE) @dataclass @@ -166,6 +166,19 @@ def describe_profile(profile_name: str, *, overwrite: bool = False, timeout: Opt raw = "" parsed = _extract_json_blob(raw) if parsed is None: + # A response that is JSON-SHAPED (starts with `{`, once code fences are stripped) but + # failed to parse is a malformed/truncated structured reply -- e.g. the aux model or + # transport cut it off mid-object -- not free-form prose. Persisting it verbatim writes + # a raw JSON fragment (literal `{`, `\n` escapes, a sentence chopped mid-word) into + # profile.yaml's description field (#104067). Only fall back to "whole reply is prose" + # when the reply never looked like JSON in the first place. + stripped = _FENCE_RE.sub("", raw.strip()) + if stripped.startswith("{"): + logger.info( + "describe: %s aux response looked JSON-shaped but failed to parse " + "(likely truncated) -- refusing to persist the raw fragment", canon, + ) + return DescribeOutcome(canon, False, "LLM returned malformed/truncated JSON response") # Fall back: raw text trimmed to one paragraph. text = raw.strip().split("\n\n", 1)[0] if not text: diff --git a/tests/hermes_cli/test_profile_describer.py b/tests/hermes_cli/test_profile_describer.py index d056fc2580..649c44da22 100644 --- a/tests/hermes_cli/test_profile_describer.py +++ b/tests/hermes_cli/test_profile_describer.py @@ -76,6 +76,103 @@ def test_describer_writes_description_with_auto_true(profile_env, monkeypatch): assert meta["description_auto"] is True +def test_describer_rejects_truncated_json_response(profile_env, monkeypatch): + # Real-world shape (#104067): the aux model's response is cut off mid-object -- no + # closing brace/quote -- so `_extract_json_blob` can't parse it. The old fallback + # persisted this raw fragment verbatim as the description; it must now be refused. + monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") + monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) + monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) + + truncated = '{\n "description": "Generalist agent that writes and debugs code, orchestrates autonomous sub-agents, and automates macOS/App' + with _patch_aux_client(truncated), patch( + "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} + ): + outcome = describer.describe_profile("myprof") + + assert outcome.ok is False + assert "malformed" in outcome.reason.lower() or "truncated" in outcome.reason.lower() + # Nothing was written: no profile.yaml, or if one exists it has no description key. + meta = profiles_mod.read_profile_meta(profile_env) + assert not meta.get("description") + + +def test_describer_rejects_json_shaped_response_with_no_closing_brace_at_all(profile_env, monkeypatch): + # Even more truncated: cut off right after the opening brace, before any key. + monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") + monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) + monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) + + with _patch_aux_client("{\n \"desc"), patch( + "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} + ): + outcome = describer.describe_profile("myprof") + + assert outcome.ok is False + meta = profiles_mod.read_profile_meta(profile_env) + assert not meta.get("description") + + +def test_describer_rejects_fenced_json_shaped_truncated_response(profile_env, monkeypatch): + # A ```json fence around a truncated object must still be recognized as JSON-shaped + # after fence stripping, not fall through to the raw-text fallback. + monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") + monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) + monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) + + fenced_truncated = '```json\n{\n "description": "Generalist agent that writes and debugs cod' + with _patch_aux_client(fenced_truncated), patch( + "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} + ): + outcome = describer.describe_profile("myprof") + + assert outcome.ok is False + meta = profiles_mod.read_profile_meta(profile_env) + assert not meta.get("description") + + +def test_describer_rejects_uppercase_fenced_json_shaped_truncated_response(profile_env, monkeypatch): + # Regression for review feedback on #104075: an uppercase ```JSON fence must be + # stripped the same as a lowercase one. Before, _FENCE_RE was case-sensitive, so + # stripping this left "JSON\n{...}" -- which doesn't start with "{" -- and the + # truncated fragment fell through to the raw-text-as-prose fallback and got persisted + # anyway, defeating the fix for this fence variant. + monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") + monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) + monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) + + fenced_truncated = '```JSON\n{\n "description": "Generalist agent that writes and debugs cod' + with _patch_aux_client(fenced_truncated), patch( + "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} + ): + outcome = describer.describe_profile("myprof") + + assert outcome.ok is False + meta = profiles_mod.read_profile_meta(profile_env) + assert not meta.get("description") + + +def test_describer_still_accepts_plain_prose_fallback(profile_env, monkeypatch): + # Regression guard: a model that ignores the JSON instruction and just replies in + # plain prose (never looked JSON-shaped) must still hit the existing lenient + # raw-text fallback -- this fix must not make the describer stricter than before + # for genuinely non-JSON replies. + monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") + monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) + monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) + + with _patch_aux_client("Writes and debugs Python codebases end to end."), patch( + "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} + ): + outcome = describer.describe_profile("myprof") + + assert outcome.ok, outcome.reason + assert outcome.description == "Writes and debugs Python codebases end to end." + meta = profiles_mod.read_profile_meta(profile_env) + assert meta["description"] == "Writes and debugs Python codebases end to end." + assert meta["description_auto"] is True + + def test_describer_refuses_to_overwrite_user_authored(profile_env, monkeypatch): profiles_mod.write_profile_meta( profile_env, description="curated", description_auto=False, From 0edab1ac4d4f72f489945edbb62a34f69e6864b3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:25:27 -0700 Subject: [PATCH 106/276] test(profiles): two invariants for the truncated-JSON refusal, comment trimmed to the why Parametrized refusal test (bare/fenced/uppercase-fenced truncated objects) asserts profile.yaml is byte-identical afterwards; one prose test keeps the lenient fallback contract. Replaces five near-duplicate tests from #104075. Campaign tracker: https://github.com/NousResearch/hermes-agent/issues/104154 --- hermes_cli/profile_describer.py | 8 +- tests/hermes_cli/test_profile_describer.py | 108 +++++---------------- 2 files changed, 25 insertions(+), 91 deletions(-) diff --git a/hermes_cli/profile_describer.py b/hermes_cli/profile_describer.py index 2e17e0b19f..31c9598c2a 100644 --- a/hermes_cli/profile_describer.py +++ b/hermes_cli/profile_describer.py @@ -166,12 +166,8 @@ def describe_profile(profile_name: str, *, overwrite: bool = False, timeout: Opt raw = "" parsed = _extract_json_blob(raw) if parsed is None: - # A response that is JSON-SHAPED (starts with `{`, once code fences are stripped) but - # failed to parse is a malformed/truncated structured reply -- e.g. the aux model or - # transport cut it off mid-object -- not free-form prose. Persisting it verbatim writes - # a raw JSON fragment (literal `{`, `\n` escapes, a sentence chopped mid-word) into - # profile.yaml's description field (#104067). Only fall back to "whole reply is prose" - # when the reply never looked like JSON in the first place. + # JSON-shaped but unparseable = the requested object got cut off (#104067); the prose + # fallback below is only for models that never attempted JSON. stripped = _FENCE_RE.sub("", raw.strip()) if stripped.startswith("{"): logger.info( diff --git a/tests/hermes_cli/test_profile_describer.py b/tests/hermes_cli/test_profile_describer.py index 649c44da22..d01d9a3128 100644 --- a/tests/hermes_cli/test_profile_describer.py +++ b/tests/hermes_cli/test_profile_describer.py @@ -76,101 +76,39 @@ def test_describer_writes_description_with_auto_true(profile_env, monkeypatch): assert meta["description_auto"] is True -def test_describer_rejects_truncated_json_response(profile_env, monkeypatch): - # Real-world shape (#104067): the aux model's response is cut off mid-object -- no - # closing brace/quote -- so `_extract_json_blob` can't parse it. The old fallback - # persisted this raw fragment verbatim as the description; it must now be refused. +@pytest.fixture +def registered_profile(profile_env, monkeypatch): monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) + return profile_env - truncated = '{\n "description": "Generalist agent that writes and debugs code, orchestrates autonomous sub-agents, and automates macOS/App' - with _patch_aux_client(truncated), patch( - "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} - ): - outcome = describer.describe_profile("myprof") +@pytest.mark.parametrize("raw", [ + '{\n "description": "Generalist agent that writes and debugs code, orchestrates autonomous sub-agents, and automates macOS/App', + '{\n "desc', + '```json\n{\n "description": "Generalist agent that writes and debugs cod', + '```JSON\n{\n "description": "Generalist agent that writes and debugs cod', +]) +def test_describer_refuses_json_shaped_reply_that_does_not_parse(registered_profile, raw): + """A reply that started as the requested JSON object but was cut off (#104067) is not prose: + it must be refused and leave profile.yaml untouched -- including behind an uppercase fence.""" + profiles_mod.write_profile_meta(registered_profile, description="previous", description_auto=True) + before = (registered_profile / "profile.yaml").read_bytes() + with _patch_aux_client(raw), patch("agent.auxiliary_client.get_auxiliary_extra_body", return_value={}): + outcome = describer.describe_profile("myprof", overwrite=True) assert outcome.ok is False - assert "malformed" in outcome.reason.lower() or "truncated" in outcome.reason.lower() - # Nothing was written: no profile.yaml, or if one exists it has no description key. - meta = profiles_mod.read_profile_meta(profile_env) - assert not meta.get("description") + assert (registered_profile / "profile.yaml").read_bytes() == before -def test_describer_rejects_json_shaped_response_with_no_closing_brace_at_all(profile_env, monkeypatch): - # Even more truncated: cut off right after the opening brace, before any key. - monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") - monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) - monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) - - with _patch_aux_client("{\n \"desc"), patch( - "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} - ): +def test_describer_still_accepts_plain_prose_fallback(registered_profile): + """A reply that never looked like JSON keeps the lenient one-paragraph prose fallback.""" + with _patch_aux_client("Writes and debugs Python codebases.\n\nSecond paragraph is dropped."), \ + patch("agent.auxiliary_client.get_auxiliary_extra_body", return_value={}): outcome = describer.describe_profile("myprof") - - assert outcome.ok is False - meta = profiles_mod.read_profile_meta(profile_env) - assert not meta.get("description") - - -def test_describer_rejects_fenced_json_shaped_truncated_response(profile_env, monkeypatch): - # A ```json fence around a truncated object must still be recognized as JSON-shaped - # after fence stripping, not fall through to the raw-text fallback. - monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") - monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) - monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) - - fenced_truncated = '```json\n{\n "description": "Generalist agent that writes and debugs cod' - with _patch_aux_client(fenced_truncated), patch( - "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} - ): - outcome = describer.describe_profile("myprof") - - assert outcome.ok is False - meta = profiles_mod.read_profile_meta(profile_env) - assert not meta.get("description") - - -def test_describer_rejects_uppercase_fenced_json_shaped_truncated_response(profile_env, monkeypatch): - # Regression for review feedback on #104075: an uppercase ```JSON fence must be - # stripped the same as a lowercase one. Before, _FENCE_RE was case-sensitive, so - # stripping this left "JSON\n{...}" -- which doesn't start with "{" -- and the - # truncated fragment fell through to the raw-text-as-prose fallback and got persisted - # anyway, defeating the fix for this fence variant. - monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") - monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) - monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) - - fenced_truncated = '```JSON\n{\n "description": "Generalist agent that writes and debugs cod' - with _patch_aux_client(fenced_truncated), patch( - "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} - ): - outcome = describer.describe_profile("myprof") - - assert outcome.ok is False - meta = profiles_mod.read_profile_meta(profile_env) - assert not meta.get("description") - - -def test_describer_still_accepts_plain_prose_fallback(profile_env, monkeypatch): - # Regression guard: a model that ignores the JSON instruction and just replies in - # plain prose (never looked JSON-shaped) must still hit the existing lenient - # raw-text fallback -- this fix must not make the describer stricter than before - # for genuinely non-JSON replies. - monkeypatch.setattr(profiles_mod, "profile_exists", lambda n: n == "myprof") - monkeypatch.setattr(profiles_mod, "normalize_profile_name", lambda n: n) - monkeypatch.setattr(profiles_mod, "get_profile_dir", lambda n: profile_env) - - with _patch_aux_client("Writes and debugs Python codebases end to end."), patch( - "agent.auxiliary_client.get_auxiliary_extra_body", return_value={} - ): - outcome = describer.describe_profile("myprof") - assert outcome.ok, outcome.reason - assert outcome.description == "Writes and debugs Python codebases end to end." - meta = profiles_mod.read_profile_meta(profile_env) - assert meta["description"] == "Writes and debugs Python codebases end to end." - assert meta["description_auto"] is True + assert outcome.description == "Writes and debugs Python codebases." + assert profiles_mod.read_profile_meta(registered_profile)["description"] == outcome.description def test_describer_refuses_to_overwrite_user_authored(profile_env, monkeypatch): From 7fc0c649ff003fbf1b44c278ac5743c78bb2592b Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Sun, 16 Aug 2026 22:24:15 +0800 Subject: [PATCH 107/276] fix(slack): drop markdown_text from native task-card appendStream payload chat.appendStream rejects a request carrying both markdown_text and chunks with `cannot_provide_both_markdown_text_and_chunks`, so every native task-card update failed and each turn silently downgraded to the plain-text fallback message (#87743). The chunks are the native rendering; the gateway caller keeps its own editable-text fallback rail for when the native call itself fails, so the markdown_text field was both rejected and redundant. Fixes #87743 --- plugins/platforms/slack/adapter.py | 5 ++-- tests/gateway/test_slack.py | 39 ++++++++++++++++++++++++++++++ 2 files changed, 42 insertions(+), 2 deletions(-) diff --git a/plugins/platforms/slack/adapter.py b/plugins/platforms/slack/adapter.py index 0bc4b94eff..5cd278a3e6 100644 --- a/plugins/platforms/slack/adapter.py +++ b/plugins/platforms/slack/adapter.py @@ -1934,8 +1934,9 @@ class SlackAdapter(BasePlatformAdapter): chunks.extend(self._task_update_chunk(task) for task in tasks) append_payload: Dict[str, Any] = { "channel": chat_id, "ts": stream.stream_ts, "chunks": chunks} - if fallback_text: - append_payload["markdown_text"] = fallback_text + # chunks-only: Slack rejects markdown_text alongside chunks + # (cannot_provide_both_markdown_text_and_chunks, #87743); the gateway owns + # the editable-text fallback rail that fallback_text feeds when this call fails. await client.api_call("chat.appendStream", json=append_payload) return SendResult(success=True, message_id=stream.stream_ts) except Exception as exc: # pragma: no cover - defensive logging diff --git a/tests/gateway/test_slack.py b/tests/gateway/test_slack.py index 055b6ed35a..5b94c36395 100644 --- a/tests/gateway/test_slack.py +++ b/tests/gateway/test_slack.py @@ -5143,6 +5143,45 @@ class TestNativeTaskCardProgress: ] assert adapter._native_task_card_streams == {} + @pytest.mark.asyncio + async def test_append_payload_never_mixes_markdown_text_with_chunks( + self, adapter + ): + """#87743: chat.appendStream rejects a request carrying both + markdown_text and chunks (`cannot_provide_both_markdown_text_and_chunks`), + which made every native task-card update fail and silently downgraded + each turn to the plain-text fallback. The fallback_text must never be + attached to the chunks payload.""" + client = adapter._app.client + + async def api_call(method, *, json): + if method == "chat.startStream": + return {"ts": "stream-1"} + return {"ok": True} + + client.api_call.side_effect = api_call + + result = await adapter.send_native_task_card_progress( + "C1", + [{"id": "call-1", "title": "terminal", "status": "in_progress"}], + metadata={"thread_id": "thread-1"}, + fallback_text="fallback progress text", + ) + + assert result.success is True + append_calls = [ + call + for call in client.api_call.await_args_list + if call.args[0] == "chat.appendStream" + ] + assert append_calls, "expected an appendStream call" + payload = append_calls[0].kwargs["json"] + assert "chunks" in payload + assert "markdown_text" not in payload, ( + "appendStream must not mix markdown_text with chunks — Slack " + "rejects the pair and the whole native card fails (#87743)" + ) + # --------------------------------------------------------------------------- # TestSlackAuthoredTextDeduplication From 4c287fa7bf479a19e925ead89cd19bd8fe96e991 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:25:00 -0700 Subject: [PATCH 108/276] test(evals): wire-contract A/B harness for Slack chat.*Stream payloads Replays the native task-card rail and the text-stream rail through the real slack_sdk AsyncWebClient against a local aiohttp receiver that enforces the documented markdown_text/chunks mutual exclusion. Local contract proof, not Slack live: records outgoing bodies before/after #87743. --- evals/slack_stream_wire_contract.py | 131 ++++++++++++++++++++++++++++ 1 file changed, 131 insertions(+) create mode 100644 evals/slack_stream_wire_contract.py diff --git a/evals/slack_stream_wire_contract.py b/evals/slack_stream_wire_contract.py new file mode 100644 index 0000000000..e194985dbb --- /dev/null +++ b/evals/slack_stream_wire_contract.py @@ -0,0 +1,131 @@ +"""Wire-contract A/B for Slack native task-card streams (#87743). + +Runs the REAL SlackAdapter.send_native_task_card_progress / stop_native_task_card_progress and +send_draft/_seal_stream paths against a REAL slack_sdk AsyncWebClient whose ``base_url`` points at a +local aiohttp receiver that records every request body. The receiver enforces Slack's documented +mutual-exclusion rule for chat.startStream/appendStream/stopStream (``markdown_text`` and +``chunks`` cannot both be present → ``cannot_provide_both_markdown_text_and_chunks``). + +THIS IS WIRE-CONTRACT PROOF, NOT SLACK LIVE: no Slack workspace or token is involved. It proves what +bytes the adapter puts on the wire and that they satisfy the documented contract. + +Usage: + evals/slack_stream_wire_contract.py --repo --out +""" +from __future__ import annotations + +import argparse +import asyncio +import json +import os +import sys +import time + +EXCLUSIVE_METHODS = {"chat.startStream", "chat.appendStream", "chat.stopStream"} + + +def _make_app(log): + from aiohttp import web + + async def handler(request): + method = request.match_info["method"] + ctype = request.headers.get("Content-Type", "") + if "json" in ctype: + body = await request.json() + else: + body = dict(await request.post()) + entry = {"method": method, "content_type": ctype, "body": body} + log.append(entry) + if method in EXCLUSIVE_METHODS and "markdown_text" in body and "chunks" in body: + entry["response"] = {"ok": False, "error": "cannot_provide_both_markdown_text_and_chunks"} + else: + entry["response"] = {"ok": True, "channel": body.get("channel"), "ts": body.get("ts") or "1700000000.000100"} + return web.json_response(entry["response"]) + + app = web.Application() + app.router.add_post("/api/{method}", handler) + return app + + +async def run(repo: str, out: str): + from aiohttp import web + + sys.path.insert(0, repo) + for m in list(sys.modules): + if m.startswith(("gateway", "plugins", "tools", "hermes", "agent")): + del sys.modules[m] + from slack_sdk.web.async_client import AsyncWebClient + from gateway.config import PlatformConfig + from plugins.platforms.slack.adapter import SlackAdapter, SLACK_AVAILABLE + + assert SLACK_AVAILABLE, "slack_sdk must be importable for a real-client wire probe" + + log: list = [] + runner = web.AppRunner(_make_app(log)) + await runner.setup() + site = web.TCPSite(runner, "127.0.0.1", 0) + await site.start() + port = site._server.sockets[0].getsockname()[1] + base_url = f"http://127.0.0.1:{port}/api/" + + adapter = SlackAdapter(PlatformConfig(enabled=True, token="xoxb-wire-probe")) + + class _App: # minimal stand-in for slack_bolt AsyncApp: only .client is read by these paths + client = AsyncWebClient(token="xoxb-wire-probe", base_url=base_url) + + adapter._app = _App() + metadata = {"thread_id": "1700000000.000001", "user_id": "U1", "recipient_team_id": "T1", "recipient_user_id": "U1"} + results = {} + + # --- Task-card rail (the bug): start + append(chunks) + stop + r1 = await adapter.send_native_task_card_progress( + "C1", [{"id": "call-1", "title": "terminal - ls", "status": "in_progress"}], + metadata=metadata, fallback_text="Hermes is working\n- terminal - ls - running") + r2 = await adapter.send_native_task_card_progress( + "C1", [{"id": "call-1", "title": "terminal - ls", "status": "complete"}], + metadata=metadata, fallback_text="Hermes is working\n- terminal - ls - complete") + await adapter.stop_native_task_card_progress("C1", metadata=metadata) + results["task_card"] = {"first": {"success": r1.success, "error": r1.error}, + "second": {"success": r2.success, "error": r2.error}} + + # --- Text stream rail (regression): start(markdown_text) + append(markdown_text) + stop(markdown_text) + d1 = await adapter.send_draft("C2", 1, "Hello", metadata=metadata) + d2 = await adapter.send_draft("C2", 1, "Hello world", metadata=metadata) + stream = adapter._active_streams.get("C2") + sealed = await adapter._seal_stream("C2", stream, final_text="Hello world!") if stream else None + results["text_stream"] = {"first": {"success": d1.success, "error": d1.error}, + "second": {"success": d2.success, "error": d2.error}, "sealed": sealed} + + await runner.cleanup() + + violations = [e for e in log if e["method"] in EXCLUSIVE_METHODS and "markdown_text" in e["body"] and "chunks" in e["body"]] + summary = { + "repo": repo, "utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()), + "label": "WIRE-CONTRACT PROOF (local receiver, slack_sdk real client) — NOT Slack live", + "slack_sdk_version": __import__("slack_sdk.version", fromlist=["__version__"]).__version__, + "requests": [{"method": e["method"], "content_type": e["content_type"], + "body_keys": sorted(e["body"].keys()), "body": e["body"], "response": e["response"]} for e in log], + "violations": [{"method": v["method"], "body_keys": sorted(v["body"].keys())} for v in violations], + "results": results, + } + with open(out, "w", encoding="utf-8") as f: + json.dump(summary, f, indent=2) + seq = [(e["method"], sorted(e["body"].keys()), e["response"].get("error")) for e in log] + for s in seq: + print(*s) + print("violations:", len(violations), "| task_card:", results["task_card"], "| text_stream:", results["text_stream"]) + return 1 if violations else 0 + + +def main(): + ap = argparse.ArgumentParser() + ap.add_argument("--repo", required=True) + ap.add_argument("--out", required=True) + a = ap.parse_args() + os.environ.setdefault("HERMES_HOME", os.path.join(os.path.dirname(a.out), "hermes-home-probe")) + os.makedirs(os.environ["HERMES_HOME"], exist_ok=True) + sys.exit(asyncio.run(run(os.path.abspath(a.repo), a.out))) + + +if __name__ == "__main__": + main() From 7ba9ac6ab75f889928d5c82f345253ad7441dbd4 Mon Sep 17 00:00:00 2001 From: Edizzier Date: Sat, 5 Sep 2026 13:23:55 +0300 Subject: [PATCH 109/276] fix(gateway): stop printing sys.maxsize as an iteration ceiling MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rebase of the existing fix onto current main: the god-file split that was in flight when this PR opened has landed, moving both original call sites out of gateway/run.py into gateway/run_busy.py and gateway/run_turn.py, which is why this PR showed a merge conflict. Also folds in a third call site (see below) that a competing PR (#102845, closed by its author in favor of this one) identified after this PR first opened. Three user-facing gateway status lines render "iteration N/M" from AIAgent.get_activity_summary()'s api_call_count / max_iterations pair: the long-running heartbeat (run_turn.GatewayTurnMixin. _run_agent_notify_long_running), the busy-session acknowledgment (run_busy.GatewayBusySessionMixin._compose_busy_ack_message), and the gateway-timeout diagnostic message shown to the user when a run is force-timed-out for inactivity (run_turn.GatewayTurnMixin. _run_agent_timeout_result). AIAgent.max_iterations defaults to sys.maxsize (unlimited tool-calling iterations for a top-level session -- see run_agent.py), so all three printed the literal 9223372036854775807 as the denominator, e.g.: ⏳ Working — 3 min — iteration 2/9223372036854775807, receiving stream response That reads as a bug rather than "unbounded" and is meaningless to a user. Add _format_iteration_progress() to gateway/run.py, a small shared formatting helper: once the configured max_iterations is at or above sys.maxsize, it renders "iteration N" alone and omits the denominator; a genuinely finite budget (e.g. a subagent's delegation. max_iterations: 250) still renders "iteration N/M" as before. All three call sites now go through it. The gateway-timeout diagnostic message's own operator-facing logger.error() call keeps the raw resolved value for diagnostics -- only the two user-facing diag_lines built from it are reformatted. Added tests/gateway/test_format_iteration_progress.py (the helper's own unit tests: unbounded default, above-sentinel values, finite budgets, and graceful handling of a missing/malformed max_iterations), tests/gateway/test_gateway_timeout_iteration_progress.py (both diagnostic-message branches, unbounded and finite, plus the heartbeat call site exercised end to end through its async polling loop), and a new regression test in tests/gateway/test_busy_session_ack.py:: TestBusySessionAck::test_status_detail_omits_denominator_for_unbounded_max_iterations that exercises the busy-ack call site end to end with the real sys.maxsize default and asserts the sentinel never reaches the rendered text. Defect 1 in the original report (a direct_result tool's raw output occasionally becoming the user-visible reply) is left for a separate change -- the reporter frames it as an open core-level design question ("a per-turn suppression option... would fix the class for all plugins"), not a drop-in fix, and it touches reply composition rather than status-line formatting. Part of #102806 (defect 2: the iteration-ceiling display; defect 1 is a separate change). --- gateway/run.py | 26 ++++ gateway/run_busy.py | 7 +- gateway/run_turn.py | 19 ++- tests/gateway/test_busy_session_ack.py | 43 +++++++ .../gateway/test_format_iteration_progress.py | 47 +++++++ ...test_gateway_timeout_iteration_progress.py | 116 ++++++++++++++++++ 6 files changed, 251 insertions(+), 7 deletions(-) create mode 100644 tests/gateway/test_format_iteration_progress.py create mode 100644 tests/gateway/test_gateway_timeout_iteration_progress.py diff --git a/gateway/run.py b/gateway/run.py index 0a62103a77..31f19f2e73 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -940,6 +940,32 @@ def _float_env(name: str, default: float) -> float: return float(default) +def _format_iteration_progress(api_call_count: Any, max_iterations: Any) -> str: + """Render an ``iteration N/M`` status fragment for a running turn. + + Shared by every user-facing status line that formats + ``AIAgent.get_activity_summary()``'s ``api_call_count`` / ``max_iterations`` pair: the + long-running heartbeat (``run_turn._run_agent_notify_long_running``), the busy-session + acknowledgment (``run_busy._handle_active_session_busy_message``), and the + gateway-timeout diagnostic message (``run_turn._run_agent_timeout_result``). + + ``AIAgent.max_iterations`` defaults to ``sys.maxsize`` (unlimited tool-calling + iterations — see ``run_agent.py``), so a top-level session almost never has a real + ceiling. Printing that default verbatim renders the literal ``9223372036854775807`` in + a user-facing status line, which reads as a bug rather than "unbounded" (#102806). Omit + the denominator once the configured ceiling is at or above that sentinel; a genuinely + finite budget (e.g. a subagent's ``delegation.max_iterations: 250``) still prints both + numbers. + """ + try: + _max = int(max_iterations) + except (TypeError, ValueError): + _max = None + if _max is None or _max >= sys.maxsize: + return f"iteration {api_call_count}" + return f"iteration {api_call_count}/{_max}" + + def _stamp_hygiene_compression_provenance( agent: Any, desc: str, provenance: "ActivityProvenance", debug_label: str) -> None: """Best-effort activity provenance stamp for hygiene compression transitions.""" diff --git a/gateway/run_busy.py b/gateway/run_busy.py index 3fcf447eb3..c3b75703b5 100644 --- a/gateway/run_busy.py +++ b/gateway/run_busy.py @@ -534,7 +534,8 @@ class GatewayBusySessionMixin: demoted_for_subagents: bool, demoted_for_compression: bool, ) -> str: from gateway.run import ( - _AGENT_PENDING_SENTINEL, _hermes_home, _load_gateway_config, _platform_config_key + _AGENT_PENDING_SENTINEL, _format_iteration_progress, _hermes_home, + _load_gateway_config, _platform_config_key, ) from gateway.display_config import resolve_display_setting @@ -556,7 +557,9 @@ class GatewayBusySessionMixin: status_parts.append(f"{elapsed_min} min elapsed") if summary.get("max_iterations", 0): status_parts.append( - f"iteration {summary.get('api_call_count', 0)}/{summary.get('max_iterations', 0)}" + _format_iteration_progress( + summary.get("api_call_count", 0), summary.get("max_iterations", 0) + ) ) if summary.get("current_tool"): status_parts.append(f"running: {summary.get('current_tool')}") diff --git a/gateway/run_turn.py b/gateway/run_turn.py index 3b7ee873fc..d5f5aacd42 100644 --- a/gateway/run_turn.py +++ b/gateway/run_turn.py @@ -3156,7 +3156,9 @@ class GatewayTurnMixin: def _run_agent_timeout_result(self, worker, turn_ctx: TurnContext) -> dict: """Synthetic failed run dict for an inactivity timeout, with the activity-tracker diagnostic; interrupts the agent if it is still running so the thread pool worker is freed.""" - from gateway.run import _INTERRUPT_REASON_TIMEOUT, request_hard_interrupt + from gateway.run import ( + _INTERRUPT_REASON_TIMEOUT, _format_iteration_progress, request_hard_interrupt, + ) session_key, result_holder, tools_holder = turn_ctx.session_key, turn_ctx.result_holder, turn_ctx.tools_holder _timed_out_agent = turn_ctx.agent_holder[0] _activity = self._agent_activity_summary(_timed_out_agent) @@ -3165,6 +3167,8 @@ class GatewayTurnMixin: _cur_tool = _activity.get("current_tool") _iter_n = _activity.get("api_call_count", 0) _iter_max = _activity.get("max_iterations", 0) + # Operator-facing log keeps the raw resolved value for diagnostics; only the two + # user-facing _diag_lines below go through _format_iteration_progress (#102806). logger.error( "Agent idle for %.0fs (timeout %.0fs) in session %s " "| last_activity=%s | iteration=%s/%s | tool=%s", @@ -3174,18 +3178,19 @@ class GatewayTurnMixin: if _timed_out_agent: request_hard_interrupt(_timed_out_agent, _INTERRUPT_REASON_TIMEOUT) _timeout_mins = int(worker.agent_timeout // 60) or 1 + _iter_progress = _format_iteration_progress(_iter_n, _iter_max) _diag_lines = [ f"⏱️ Agent inactive for {_timeout_mins} min — no tool calls or API responses." ] if _cur_tool: _diag_lines.append( f"The agent appears stuck on tool `{_cur_tool}` ({_secs_ago:.0f}s since last " - f"activity, iteration {_iter_n}/{_iter_max})." + f"activity, {_iter_progress})." ) else: _diag_lines.append( f"Last activity: {_last_desc} ({_secs_ago:.0f}s ago, " - f"iteration {_iter_n}/{_iter_max}). " + f"{_iter_progress}). " "The agent may have been waiting on an API response." ) _diag_lines.append( @@ -3711,7 +3716,9 @@ class GatewayTurnMixin: Interval: agent.gateway_notify_interval / HERMES_AGENT_NOTIFY_INTERVAL (default 180s; 0 or long_running_notifications=off disables).""" - from gateway.run import _float_env, _interim_metadata, _non_conversational_metadata + from gateway.run import ( + _float_env, _format_iteration_progress, _interim_metadata, _non_conversational_metadata, + ) _notify_start = time.time() _NOTIFY_INTERVAL = _float_env("HERMES_AGENT_NOTIFY_INTERVAL", 180) _long_running_mode = disp._display_surface_mode("long_running_notifications", default=True, allow_generic=True) @@ -3740,7 +3747,9 @@ class GatewayTurnMixin: if _a: _parts = [] if _want_iteration_detail: - _parts.append(f"iteration {_a['api_call_count']}/{_a['max_iterations']}") + _parts.append( + _format_iteration_progress(_a["api_call_count"], _a["max_iterations"]) + ) _action = _a.get("current_tool") or _a.get("last_activity_desc") if _action: _parts.append(str(_action)) diff --git a/tests/gateway/test_busy_session_ack.py b/tests/gateway/test_busy_session_ack.py index fc7beb9aa1..6c4ce15778 100644 --- a/tests/gateway/test_busy_session_ack.py +++ b/tests/gateway/test_busy_session_ack.py @@ -402,6 +402,49 @@ class TestBusySessionAck: assert "terminal" in content # current tool assert "10 min" in content # elapsed + @pytest.mark.asyncio + async def test_status_detail_omits_denominator_for_unbounded_max_iterations( + self, monkeypatch, + ): + """#102806: a top-level session's real max_iterations is sys.maxsize + (unlimited — see AIAgent's default). The busy-ack must not print that + literal sentinel as an iteration ceiling.""" + import sys + + import gateway.run as _gr + + monkeypatch.setattr( + _gr, + "_load_gateway_config", + lambda: {"display": {"platforms": {"telegram": {"busy_ack_detail": True}}}}, + ) + runner, sentinel = _make_runner() + runner._busy_input_mode = "interrupt" + adapter = _make_adapter() + + event = _make_event(text="yo") + sk = build_session_key(event.source) + + agent = MagicMock() + agent.get_activity_summary.return_value = { + "api_call_count": 3, + "max_iterations": sys.maxsize, + "current_tool": "terminal", + "last_activity_ts": time.time(), + "last_activity_desc": "terminal", + "seconds_since_activity": 0.5, + } + runner._running_agents[sk] = agent + runner._running_agents_ts[sk] = time.time() - 600 + runner.adapters[event.source.platform] = adapter + + await runner._handle_active_session_busy_message(event, sk) + + call_kwargs = adapter._send_with_retry.call_args + content = call_kwargs.kwargs.get("content", "") + assert "iteration 3" in content + assert str(sys.maxsize) not in content + class TestBusySessionOnboardingHint: """First-touch hint appended to the busy-ack the first time it fires.""" diff --git a/tests/gateway/test_format_iteration_progress.py b/tests/gateway/test_format_iteration_progress.py new file mode 100644 index 0000000000..bad9c20b39 --- /dev/null +++ b/tests/gateway/test_format_iteration_progress.py @@ -0,0 +1,47 @@ +"""Unit tests for gateway.run._format_iteration_progress (#102806). + +``AIAgent.max_iterations`` defaults to ``sys.maxsize`` (unlimited +tool-calling iterations for a top-level session — see ``run_agent.py``). +Three user-facing gateway status lines render ``iteration N/M`` from +``AIAgent.get_activity_summary()``'s ``api_call_count`` / ``max_iterations`` +pair: the long-running heartbeat, the busy-session acknowledgment, and the +gateway-timeout diagnostic message. All three share this helper; see +``test_busy_session_ack.py`` and ``test_gateway_timeout_iteration_progress.py`` +for the call sites' own end-to-end regression tests. +""" + +import sys + +from gateway.run import _format_iteration_progress + + +class TestFormatIterationProgress: + def test_unbounded_default_omits_denominator(self): + """sys.maxsize (AIAgent's actual default) prints iteration count alone.""" + assert _format_iteration_progress(2, sys.maxsize) == "iteration 2" + + def test_above_sys_maxsize_also_omits_denominator(self): + """Anything at or beyond the sentinel is treated as unbounded too.""" + assert _format_iteration_progress(5, sys.maxsize + 1) == "iteration 5" + + def test_finite_budget_still_shows_both_numbers(self): + """A real, finite ceiling (e.g. a subagent's max_iterations: 250) + keeps the existing N/M format — this is not a blanket format change.""" + assert _format_iteration_progress(7, 250) == "iteration 7/250" + + def test_small_finite_budget(self): + assert _format_iteration_progress(0, 90) == "iteration 0/90" + + def test_non_numeric_max_iterations_falls_back_to_unbounded_rendering(self): + """get_activity_summary() is a best-effort diagnostics snapshot; a + malformed/missing max_iterations must not raise inside a status-line + formatter (call sites wrap this in try/except, but the helper itself + should degrade gracefully rather than propagate a TypeError).""" + assert _format_iteration_progress(3, None) == "iteration 3" + assert _format_iteration_progress(3, "not-a-number") == "iteration 3" + + def test_api_call_count_is_rendered_verbatim(self): + """Only the denominator's unbounded-ness is special-cased; the + numerator (api_call_count) always prints as given.""" + assert _format_iteration_progress(123, 250) == "iteration 123/250" + assert _format_iteration_progress(123, sys.maxsize) == "iteration 123" diff --git a/tests/gateway/test_gateway_timeout_iteration_progress.py b/tests/gateway/test_gateway_timeout_iteration_progress.py new file mode 100644 index 0000000000..5cf5f590ad --- /dev/null +++ b/tests/gateway/test_gateway_timeout_iteration_progress.py @@ -0,0 +1,116 @@ +"""#102806: the gateway inactivity-timeout diagnostic message and the +long-running heartbeat -- the two remaining user-facing render sites +alongside the busy-ack (see test_busy_session_ack.py) -- must not print +sys.maxsize as an iteration ceiling either. + +``_run_agent_timeout_result`` builds the synthetic ``final_response`` shown +to the user when an agent run is force-timed-out for inactivity. It embeds +``iteration N/M`` twice (the "stuck on tool" branch and the "last activity" +branch). ``_run_agent_notify_long_running`` embeds it once, in the periodic +"still working" heartbeat. All three call sites (this file's two classes, +plus test_busy_session_ack.py's busy-ack test) share the same +``_format_iteration_progress`` helper; see test_format_iteration_progress.py +for the helper's own unit tests. +""" + +from __future__ import annotations + +import sys +from types import SimpleNamespace +from unittest.mock import AsyncMock, MagicMock + +import pytest + +from gateway.run_turn import GatewayTurnMixin +from gateway.turn_context import TurnContext + + +def _make_worker(agent_timeout=1800.0): + return SimpleNamespace(agent_timeout=agent_timeout) + + +def _make_turn_ctx(agent): + return TurnContext(session_key="sess-1", agent_holder=[agent]) + + +class TestTimeoutDiagnosticIterationProgress: + def test_unbounded_max_iterations_omits_denominator_when_stuck_on_tool(self, monkeypatch): + mixin = GatewayTurnMixin() + monkeypatch.setattr("gateway.run.request_hard_interrupt", MagicMock(), raising=False) + agent = MagicMock() + agent.get_activity_summary.return_value = { + "last_activity_desc": "tool_call", + "seconds_since_activity": 42.0, + "current_tool": "terminal", + "api_call_count": 5, + "max_iterations": sys.maxsize, + } + result = mixin._run_agent_timeout_result(_make_worker(), _make_turn_ctx(agent)) + assert "iteration 5" in result["final_response"] + assert str(sys.maxsize) not in result["final_response"] + + def test_unbounded_max_iterations_omits_denominator_in_last_activity_branch(self, monkeypatch): + mixin = GatewayTurnMixin() + monkeypatch.setattr("gateway.run.request_hard_interrupt", MagicMock(), raising=False) + agent = MagicMock() + agent.get_activity_summary.return_value = { + "last_activity_desc": "api_call_streaming", + "seconds_since_activity": 12.0, + "current_tool": None, + "api_call_count": 2, + "max_iterations": sys.maxsize, + } + result = mixin._run_agent_timeout_result(_make_worker(), _make_turn_ctx(agent)) + assert "iteration 2" in result["final_response"] + assert str(sys.maxsize) not in result["final_response"] + + def test_finite_max_iterations_still_shows_both_numbers(self, monkeypatch): + mixin = GatewayTurnMixin() + monkeypatch.setattr("gateway.run.request_hard_interrupt", MagicMock(), raising=False) + agent = MagicMock() + agent.get_activity_summary.return_value = { + "last_activity_desc": "tool_call", + "seconds_since_activity": 5.0, + "current_tool": "code_exec", + "api_call_count": 7, + "max_iterations": 250, + } + result = mixin._run_agent_timeout_result(_make_worker(), _make_turn_ctx(agent)) + assert "iteration 7/250" in result["final_response"] + + +class TestLongRunningHeartbeatIterationProgress: + """The heartbeat's own render call, exercised end to end through its + async polling loop (one iteration, then the loop is told to stop).""" + + @pytest.mark.asyncio + async def test_heartbeat_omits_denominator_for_unbounded_max_iterations(self, monkeypatch): + monkeypatch.setenv("HERMES_AGENT_NOTIFY_INTERVAL", "0.01") + + mixin = GatewayTurnMixin() + adapter = MagicMock() + adapter.send = AsyncMock(return_value=SimpleNamespace(success=True, message_id="m1")) + mixin._adapter_for_source = MagicMock(return_value=adapter) + mixin._should_emit_long_running_notification = MagicMock(side_effect=[True, False]) + + agent = MagicMock() + agent.get_activity_summary.return_value = { + "api_call_count": 4, + "max_iterations": sys.maxsize, + "current_tool": "terminal", + } + + disp = MagicMock() + disp._display_surface_mode.return_value = "on" + disp.resolve_display_setting.return_value = True + + turn_ctx = TurnContext( + source=SimpleNamespace(chat_id="c1", platform="telegram"), + session_key="sess-1", agent_holder=[agent], + ) + + await mixin._run_agent_notify_long_running(disp, turn_ctx, [None]) + + sent_text = adapter.send.await_args.args[1] + assert "iteration 4" in sent_text + assert str(sys.maxsize) not in sent_text From 62f353eed56c841842a764940a786030dfbd573e Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:26:26 -0700 Subject: [PATCH 110/276] refactor(gateway): iteration-progress formatter lives in agent/session_activity, tests trimmed to the invariant bar The salvaged helper was appended to the gateway/run.py facade; new behaviour belongs in a topical sibling. agent/session_activity.py already owns the activity-snapshot contract the three render sites read from, so the formatter moves there as format_iteration_progress and the three call sites import it at module level instead of late-importing the facade. Tests trimmed to the salvage bar (<= 2 invariant tests per fix): one parameterized contract on the helper (unbounded/None hide the ceiling, a real budget keeps N/M) plus the contributor's end-to-end busy-ack test through the real render path. The per-site heartbeat/timeout tests exercised the same helper through mocks and are dropped; evals/gateway_status_render/ iteration_ceiling_ab.py drives all three real render sites with a real AIAgent for before/after evidence (3 sentinel leaks on main -> 0). Related: #103109 (same fix, same target module; credited in the PR), #102845 (third render site, closed by its author in favour of #102817). --- agent/session_activity.py | 16 +++ .../iteration_ceiling_ab.py | 115 +++++++++++++++++ gateway/run.py | 26 ---- gateway/run_busy.py | 6 +- gateway/run_turn.py | 18 +-- tests/agent/test_session_activity.py | 18 +++ .../gateway/test_format_iteration_progress.py | 47 ------- ...test_gateway_timeout_iteration_progress.py | 116 ------------------ 8 files changed, 158 insertions(+), 204 deletions(-) create mode 100644 evals/gateway_status_render/iteration_ceiling_ab.py delete mode 100644 tests/gateway/test_format_iteration_progress.py delete mode 100644 tests/gateway/test_gateway_timeout_iteration_progress.py diff --git a/agent/session_activity.py b/agent/session_activity.py index ddd04c3ab7..1ab19ef978 100644 --- a/agent/session_activity.py +++ b/agent/session_activity.py @@ -4,6 +4,7 @@ only (notification, timeout, kill and retry policy live elsewhere). Provenance i from __future__ import annotations +import sys import time from contextlib import suppress from enum import Enum @@ -45,6 +46,21 @@ def normalize_activity_provenance(provenance: Optional[ActivityProvenance | str] return ActivityProvenance.UNKNOWN +def format_iteration_progress(api_call_count: Any, max_iterations: Any) -> str: + """``iteration N/M`` for user-facing status lines, or ``iteration N`` when the cap is unbounded. + + ``AIAgent.max_iterations`` defaults to ``sys.maxsize`` (unlimited), so printing the pair verbatim + shows ``iteration 3/9223372036854775807`` in busy acks, heartbeats and timeout diagnostics (#102806). + """ + try: + cap = int(max_iterations) + except (TypeError, ValueError): + cap = sys.maxsize + if cap >= sys.maxsize: + return f"iteration {api_call_count}" + return f"iteration {api_call_count}/{cap}" + + def reset_session_activity_persist_window(agent: Any) -> None: """Clear the persist rate-limit so the next stamp writes through (terminal compression labels must not stick on mid-compress text).""" with suppress(Exception): diff --git a/evals/gateway_status_render/iteration_ceiling_ab.py b/evals/gateway_status_render/iteration_ceiling_ab.py new file mode 100644 index 0000000000..8c4dabcb06 --- /dev/null +++ b/evals/gateway_status_render/iteration_ceiling_ab.py @@ -0,0 +1,115 @@ +"""A/B probe: what the three user-facing gateway status lines render for the iteration counter. + +Drives the REAL render paths (busy-ack, long-running heartbeat, inactivity-timeout diagnostic) +with a REAL ``AIAgent`` constructed with its default (unlimited) ``max_iterations`` and a +stub adapter that records the outbound text. Run on origin/main and on the fix branch: + + HERMES_HOME=$(mktemp -d) python evals/gateway_status_render/iteration_ceiling_ab.py [--finite 250] + +Prints one JSON object per render site with the exact text a user would see. +""" + +from __future__ import annotations + +import argparse +import asyncio +import json +import os +import sys +import time +from types import SimpleNamespace +from typing import Any +from unittest.mock import AsyncMock, MagicMock + + +def _agent(max_iterations: int | None): + from run_agent import AIAgent + + kwargs: dict[str, Any] = dict(base_url="http://127.0.0.1:9/v1", api_key="sk-dummy", model="dummy-model", + quiet_mode=True, enabled_toolsets=[], disabled_toolsets=["*"]) + if max_iterations is not None: + kwargs["max_iterations"] = max_iterations + agent = AIAgent(**kwargs) + agent._api_call_count = 3 + agent._current_tool = "terminal" + agent._last_activity_desc = "terminal" + agent._last_activity_ts = time.time() - 42 + return agent + + +async def _busy_ack(agent) -> str: + import gateway.run as gr + from gateway.platforms.base import MessageEvent, MessageType, SessionSource, build_session_key + + gr._load_gateway_config = lambda: {"display": {"platforms": {"telegram": {"busy_ack_detail": True}}}} + runner = object.__new__(gr.GatewayRunner) + runner._running_agents, runner._running_agents_ts = {}, {} + runner._pending_messages, runner._busy_ack_ts, runner._queued_events = {}, {}, {} + runner._draining, runner._busy_text_mode, runner._busy_input_mode = False, "interrupt", "interrupt" + runner.adapters, runner.config, runner.session_store = {}, MagicMock(), None + runner.config.group_sessions_per_user, runner.config.thread_sessions_per_user = True, False + runner.hooks = MagicMock(); runner.hooks.emit = AsyncMock() + runner.pairing_store = MagicMock(); runner.pairing_store.is_approved.return_value = True + runner._is_user_authorized = lambda _s: True + source = SessionSource(platform=MagicMock(value="telegram"), chat_id="123", chat_type="private", user_id="u1") + event = MessageEvent(text="status?", message_type=MessageType.TEXT, source=source, message_id="m1") + sk = build_session_key(source) + adapter = MagicMock() + adapter._pending_messages, adapter._text_debounce, adapter._busy_text_debounce_seconds = {}, {}, 0.6 + adapter._send_with_retry = AsyncMock() + adapter.config = MagicMock(); adapter.config.extra = {} + adapter.platform = MagicMock(value="telegram") + runner._running_agents[sk] = agent + runner._running_agents_ts[sk] = time.time() - 600 + runner.adapters[source.platform] = adapter + await runner._handle_active_session_busy_message(event, sk) + return adapter._send_with_retry.call_args.kwargs.get("content", "") + + +async def _heartbeat(agent) -> str: + from gateway.run_turn import GatewayTurnMixin + from gateway.turn_context import TurnContext + + os.environ["HERMES_AGENT_NOTIFY_INTERVAL"] = "0.01" + mixin = GatewayTurnMixin() + adapter = MagicMock() + adapter.send = AsyncMock(return_value=SimpleNamespace(success=True, message_id="hb1")) + mixin._adapter_for_source = MagicMock(return_value=adapter) + mixin._should_emit_long_running_notification = MagicMock(side_effect=[True, False]) + disp = MagicMock() + disp._display_surface_mode.return_value = "on" + disp.resolve_display_setting.return_value = True + ctx = TurnContext(source=SimpleNamespace(chat_id="c1", platform="telegram"), session_key="s1", agent_holder=[agent]) + await mixin._run_agent_notify_long_running(disp, ctx, [None]) + return adapter.send.await_args.args[1] + + +def _timeout(agent) -> str: + import gateway.run as gr + from gateway.run_turn import GatewayTurnMixin + from gateway.turn_context import TurnContext + + gr.request_hard_interrupt = MagicMock() + ctx = TurnContext(session_key="s1", agent_holder=[agent]) + return GatewayTurnMixin()._run_agent_timeout_result(SimpleNamespace(agent_timeout=1800.0), ctx)["final_response"] + + +def main() -> None: + ap = argparse.ArgumentParser() + ap.add_argument("--finite", type=int, default=None, help="use a finite max_iterations instead of the default") + args = ap.parse_args() + sys.path.insert(0, os.getcwd()) + agent = _agent(args.finite) + out = { + "head": os.popen("git rev-parse --short HEAD").read().strip(), + "max_iterations": agent.max_iterations, + "busy_ack": asyncio.run(_busy_ack(agent)), + "heartbeat": asyncio.run(_heartbeat(agent)), + "timeout_diag": _timeout(agent), + } + out["sentinel_leaks"] = sum(str(sys.maxsize) in v for k, v in out.items() if k in ("busy_ack", "heartbeat", "timeout_diag")) + print(json.dumps(out, indent=2, ensure_ascii=False)) + + +if __name__ == "__main__": + main() diff --git a/gateway/run.py b/gateway/run.py index 31f19f2e73..0a62103a77 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -940,32 +940,6 @@ def _float_env(name: str, default: float) -> float: return float(default) -def _format_iteration_progress(api_call_count: Any, max_iterations: Any) -> str: - """Render an ``iteration N/M`` status fragment for a running turn. - - Shared by every user-facing status line that formats - ``AIAgent.get_activity_summary()``'s ``api_call_count`` / ``max_iterations`` pair: the - long-running heartbeat (``run_turn._run_agent_notify_long_running``), the busy-session - acknowledgment (``run_busy._handle_active_session_busy_message``), and the - gateway-timeout diagnostic message (``run_turn._run_agent_timeout_result``). - - ``AIAgent.max_iterations`` defaults to ``sys.maxsize`` (unlimited tool-calling - iterations — see ``run_agent.py``), so a top-level session almost never has a real - ceiling. Printing that default verbatim renders the literal ``9223372036854775807`` in - a user-facing status line, which reads as a bug rather than "unbounded" (#102806). Omit - the denominator once the configured ceiling is at or above that sentinel; a genuinely - finite budget (e.g. a subagent's ``delegation.max_iterations: 250``) still prints both - numbers. - """ - try: - _max = int(max_iterations) - except (TypeError, ValueError): - _max = None - if _max is None or _max >= sys.maxsize: - return f"iteration {api_call_count}" - return f"iteration {api_call_count}/{_max}" - - def _stamp_hygiene_compression_provenance( agent: Any, desc: str, provenance: "ActivityProvenance", debug_label: str) -> None: """Best-effort activity provenance stamp for hygiene compression transitions.""" diff --git a/gateway/run_busy.py b/gateway/run_busy.py index c3b75703b5..0154fe2805 100644 --- a/gateway/run_busy.py +++ b/gateway/run_busy.py @@ -15,6 +15,7 @@ import json import os import time from agent.i18n import t +from agent.session_activity import format_iteration_progress from gateway.config import Platform from gateway.platforms.base import EphemeralReply, MessageEvent, MessageType from gateway.session import SessionSource @@ -534,8 +535,7 @@ class GatewayBusySessionMixin: demoted_for_subagents: bool, demoted_for_compression: bool, ) -> str: from gateway.run import ( - _AGENT_PENDING_SENTINEL, _format_iteration_progress, _hermes_home, - _load_gateway_config, _platform_config_key, + _AGENT_PENDING_SENTINEL, _hermes_home, _load_gateway_config, _platform_config_key ) from gateway.display_config import resolve_display_setting @@ -557,7 +557,7 @@ class GatewayBusySessionMixin: status_parts.append(f"{elapsed_min} min elapsed") if summary.get("max_iterations", 0): status_parts.append( - _format_iteration_progress( + format_iteration_progress( summary.get("api_call_count", 0), summary.get("max_iterations", 0) ) ) diff --git a/gateway/run_turn.py b/gateway/run_turn.py index d5f5aacd42..53684bf60e 100644 --- a/gateway/run_turn.py +++ b/gateway/run_turn.py @@ -16,6 +16,7 @@ import queue import threading import time from agent.i18n import t +from agent.session_activity import format_iteration_progress from contextlib import nullcontext, suppress from contextvars import copy_context from gateway.config import Platform @@ -3156,9 +3157,7 @@ class GatewayTurnMixin: def _run_agent_timeout_result(self, worker, turn_ctx: TurnContext) -> dict: """Synthetic failed run dict for an inactivity timeout, with the activity-tracker diagnostic; interrupts the agent if it is still running so the thread pool worker is freed.""" - from gateway.run import ( - _INTERRUPT_REASON_TIMEOUT, _format_iteration_progress, request_hard_interrupt, - ) + from gateway.run import _INTERRUPT_REASON_TIMEOUT, request_hard_interrupt session_key, result_holder, tools_holder = turn_ctx.session_key, turn_ctx.result_holder, turn_ctx.tools_holder _timed_out_agent = turn_ctx.agent_holder[0] _activity = self._agent_activity_summary(_timed_out_agent) @@ -3167,8 +3166,7 @@ class GatewayTurnMixin: _cur_tool = _activity.get("current_tool") _iter_n = _activity.get("api_call_count", 0) _iter_max = _activity.get("max_iterations", 0) - # Operator-facing log keeps the raw resolved value for diagnostics; only the two - # user-facing _diag_lines below go through _format_iteration_progress (#102806). + # Operator-facing log keeps the raw resolved value; only the user-facing lines hide the sentinel. logger.error( "Agent idle for %.0fs (timeout %.0fs) in session %s " "| last_activity=%s | iteration=%s/%s | tool=%s", @@ -3178,7 +3176,7 @@ class GatewayTurnMixin: if _timed_out_agent: request_hard_interrupt(_timed_out_agent, _INTERRUPT_REASON_TIMEOUT) _timeout_mins = int(worker.agent_timeout // 60) or 1 - _iter_progress = _format_iteration_progress(_iter_n, _iter_max) + _iter_progress = format_iteration_progress(_iter_n, _iter_max) _diag_lines = [ f"⏱️ Agent inactive for {_timeout_mins} min — no tool calls or API responses." ] @@ -3716,9 +3714,7 @@ class GatewayTurnMixin: Interval: agent.gateway_notify_interval / HERMES_AGENT_NOTIFY_INTERVAL (default 180s; 0 or long_running_notifications=off disables).""" - from gateway.run import ( - _float_env, _format_iteration_progress, _interim_metadata, _non_conversational_metadata, - ) + from gateway.run import _float_env, _interim_metadata, _non_conversational_metadata _notify_start = time.time() _NOTIFY_INTERVAL = _float_env("HERMES_AGENT_NOTIFY_INTERVAL", 180) _long_running_mode = disp._display_surface_mode("long_running_notifications", default=True, allow_generic=True) @@ -3747,9 +3743,7 @@ class GatewayTurnMixin: if _a: _parts = [] if _want_iteration_detail: - _parts.append( - _format_iteration_progress(_a["api_call_count"], _a["max_iterations"]) - ) + _parts.append(format_iteration_progress(_a["api_call_count"], _a["max_iterations"])) _action = _a.get("current_tool") or _a.get("last_activity_desc") if _action: _parts.append(str(_action)) diff --git a/tests/agent/test_session_activity.py b/tests/agent/test_session_activity.py index 7d6726a0f1..0a67442e4c 100644 --- a/tests/agent/test_session_activity.py +++ b/tests/agent/test_session_activity.py @@ -1,17 +1,35 @@ """Unit tests for the shared session activity observation contract.""" +import sys from types import SimpleNamespace +import pytest + from agent.session_activity import ( ACTIVITY_DESCRIPTION_MAX, ActivityProvenance, bound_activity_description, build_activity_snapshot, + format_iteration_progress, normalize_activity_provenance, reset_session_activity_persist_window, ) +@pytest.mark.parametrize( + "max_iterations, expected", + [ + (sys.maxsize, "iteration 3"), # AIAgent's default: unbounded, so no ceiling is shown + (None, "iteration 3"), + (250, "iteration 3/250"), # a real budget (e.g. delegation.max_iterations) keeps N/M + ], +) +def test_format_iteration_progress_hides_unbounded_ceiling(max_iterations, expected): + out = format_iteration_progress(3, max_iterations) + assert out == expected + assert str(sys.maxsize) not in out + + def test_bound_activity_description_truncates(): long = "x" * (ACTIVITY_DESCRIPTION_MAX + 80) out = bound_activity_description(long) diff --git a/tests/gateway/test_format_iteration_progress.py b/tests/gateway/test_format_iteration_progress.py deleted file mode 100644 index bad9c20b39..0000000000 --- a/tests/gateway/test_format_iteration_progress.py +++ /dev/null @@ -1,47 +0,0 @@ -"""Unit tests for gateway.run._format_iteration_progress (#102806). - -``AIAgent.max_iterations`` defaults to ``sys.maxsize`` (unlimited -tool-calling iterations for a top-level session — see ``run_agent.py``). -Three user-facing gateway status lines render ``iteration N/M`` from -``AIAgent.get_activity_summary()``'s ``api_call_count`` / ``max_iterations`` -pair: the long-running heartbeat, the busy-session acknowledgment, and the -gateway-timeout diagnostic message. All three share this helper; see -``test_busy_session_ack.py`` and ``test_gateway_timeout_iteration_progress.py`` -for the call sites' own end-to-end regression tests. -""" - -import sys - -from gateway.run import _format_iteration_progress - - -class TestFormatIterationProgress: - def test_unbounded_default_omits_denominator(self): - """sys.maxsize (AIAgent's actual default) prints iteration count alone.""" - assert _format_iteration_progress(2, sys.maxsize) == "iteration 2" - - def test_above_sys_maxsize_also_omits_denominator(self): - """Anything at or beyond the sentinel is treated as unbounded too.""" - assert _format_iteration_progress(5, sys.maxsize + 1) == "iteration 5" - - def test_finite_budget_still_shows_both_numbers(self): - """A real, finite ceiling (e.g. a subagent's max_iterations: 250) - keeps the existing N/M format — this is not a blanket format change.""" - assert _format_iteration_progress(7, 250) == "iteration 7/250" - - def test_small_finite_budget(self): - assert _format_iteration_progress(0, 90) == "iteration 0/90" - - def test_non_numeric_max_iterations_falls_back_to_unbounded_rendering(self): - """get_activity_summary() is a best-effort diagnostics snapshot; a - malformed/missing max_iterations must not raise inside a status-line - formatter (call sites wrap this in try/except, but the helper itself - should degrade gracefully rather than propagate a TypeError).""" - assert _format_iteration_progress(3, None) == "iteration 3" - assert _format_iteration_progress(3, "not-a-number") == "iteration 3" - - def test_api_call_count_is_rendered_verbatim(self): - """Only the denominator's unbounded-ness is special-cased; the - numerator (api_call_count) always prints as given.""" - assert _format_iteration_progress(123, 250) == "iteration 123/250" - assert _format_iteration_progress(123, sys.maxsize) == "iteration 123" diff --git a/tests/gateway/test_gateway_timeout_iteration_progress.py b/tests/gateway/test_gateway_timeout_iteration_progress.py deleted file mode 100644 index 5cf5f590ad..0000000000 --- a/tests/gateway/test_gateway_timeout_iteration_progress.py +++ /dev/null @@ -1,116 +0,0 @@ -"""#102806: the gateway inactivity-timeout diagnostic message and the -long-running heartbeat -- the two remaining user-facing render sites -alongside the busy-ack (see test_busy_session_ack.py) -- must not print -sys.maxsize as an iteration ceiling either. - -``_run_agent_timeout_result`` builds the synthetic ``final_response`` shown -to the user when an agent run is force-timed-out for inactivity. It embeds -``iteration N/M`` twice (the "stuck on tool" branch and the "last activity" -branch). ``_run_agent_notify_long_running`` embeds it once, in the periodic -"still working" heartbeat. All three call sites (this file's two classes, -plus test_busy_session_ack.py's busy-ack test) share the same -``_format_iteration_progress`` helper; see test_format_iteration_progress.py -for the helper's own unit tests. -""" - -from __future__ import annotations - -import sys -from types import SimpleNamespace -from unittest.mock import AsyncMock, MagicMock - -import pytest - -from gateway.run_turn import GatewayTurnMixin -from gateway.turn_context import TurnContext - - -def _make_worker(agent_timeout=1800.0): - return SimpleNamespace(agent_timeout=agent_timeout) - - -def _make_turn_ctx(agent): - return TurnContext(session_key="sess-1", agent_holder=[agent]) - - -class TestTimeoutDiagnosticIterationProgress: - def test_unbounded_max_iterations_omits_denominator_when_stuck_on_tool(self, monkeypatch): - mixin = GatewayTurnMixin() - monkeypatch.setattr("gateway.run.request_hard_interrupt", MagicMock(), raising=False) - agent = MagicMock() - agent.get_activity_summary.return_value = { - "last_activity_desc": "tool_call", - "seconds_since_activity": 42.0, - "current_tool": "terminal", - "api_call_count": 5, - "max_iterations": sys.maxsize, - } - result = mixin._run_agent_timeout_result(_make_worker(), _make_turn_ctx(agent)) - assert "iteration 5" in result["final_response"] - assert str(sys.maxsize) not in result["final_response"] - - def test_unbounded_max_iterations_omits_denominator_in_last_activity_branch(self, monkeypatch): - mixin = GatewayTurnMixin() - monkeypatch.setattr("gateway.run.request_hard_interrupt", MagicMock(), raising=False) - agent = MagicMock() - agent.get_activity_summary.return_value = { - "last_activity_desc": "api_call_streaming", - "seconds_since_activity": 12.0, - "current_tool": None, - "api_call_count": 2, - "max_iterations": sys.maxsize, - } - result = mixin._run_agent_timeout_result(_make_worker(), _make_turn_ctx(agent)) - assert "iteration 2" in result["final_response"] - assert str(sys.maxsize) not in result["final_response"] - - def test_finite_max_iterations_still_shows_both_numbers(self, monkeypatch): - mixin = GatewayTurnMixin() - monkeypatch.setattr("gateway.run.request_hard_interrupt", MagicMock(), raising=False) - agent = MagicMock() - agent.get_activity_summary.return_value = { - "last_activity_desc": "tool_call", - "seconds_since_activity": 5.0, - "current_tool": "code_exec", - "api_call_count": 7, - "max_iterations": 250, - } - result = mixin._run_agent_timeout_result(_make_worker(), _make_turn_ctx(agent)) - assert "iteration 7/250" in result["final_response"] - - -class TestLongRunningHeartbeatIterationProgress: - """The heartbeat's own render call, exercised end to end through its - async polling loop (one iteration, then the loop is told to stop).""" - - @pytest.mark.asyncio - async def test_heartbeat_omits_denominator_for_unbounded_max_iterations(self, monkeypatch): - monkeypatch.setenv("HERMES_AGENT_NOTIFY_INTERVAL", "0.01") - - mixin = GatewayTurnMixin() - adapter = MagicMock() - adapter.send = AsyncMock(return_value=SimpleNamespace(success=True, message_id="m1")) - mixin._adapter_for_source = MagicMock(return_value=adapter) - mixin._should_emit_long_running_notification = MagicMock(side_effect=[True, False]) - - agent = MagicMock() - agent.get_activity_summary.return_value = { - "api_call_count": 4, - "max_iterations": sys.maxsize, - "current_tool": "terminal", - } - - disp = MagicMock() - disp._display_surface_mode.return_value = "on" - disp.resolve_display_setting.return_value = True - - turn_ctx = TurnContext( - source=SimpleNamespace(chat_id="c1", platform="telegram"), - session_key="sess-1", agent_holder=[agent], - ) - - await mixin._run_agent_notify_long_running(disp, turn_ctx, [None]) - - sent_text = adapter.send.await_args.args[1] - assert "iteration 4" in sent_text - assert str(sys.maxsize) not in sent_text From cb3b5bb64c458ba3c8effbf408ec26e9595b8b3c Mon Sep 17 00:00:00 2001 From: HwangJohn Date: Wed, 17 Jun 2026 17:31:01 +0900 Subject: [PATCH 111/276] feat(webhook): accept standard webhook signatures --- gateway/platforms/webhook.py | 11 +++++++---- website/docs/user-guide/messaging/webhooks.md | 5 +++-- 2 files changed, 10 insertions(+), 6 deletions(-) diff --git a/gateway/platforms/webhook.py b/gateway/platforms/webhook.py index 9c928a474c..7eb63cd505 100644 --- a/gateway/platforms/webhook.py +++ b/gateway/platforms/webhook.py @@ -548,8 +548,8 @@ class WebhookAdapter(BasePlatformAdapter): prompt = self._render_prompt(route_config.get("prompt", ""), payload, event_type, route_name) if skills := route_config.get("skills", []): prompt = self._apply_skills(prompt, skills) - delivery_id = headers.get("X-GitHub-Delivery", headers.get( - "svix-id", headers.get("X-Request-ID", str(int(time.time() * 1000))))) + delivery_id = headers.get("X-GitHub-Delivery", headers.get("svix-id", headers.get( + "webhook-id", headers.get("X-Request-ID", str(int(time.time() * 1000)))))) now = time.time() # idempotency: skip duplicate deliveries (webhook retries) if not self._record_delivery_id(delivery_id, now): logger.info("[webhook] Skipping duplicate delivery %s", delivery_id) @@ -617,14 +617,17 @@ class WebhookAdapter(BasePlatformAdapter): # --- Signature validation --- def _validate_signature(self, request: "web.Request", body: bytes, secret: str) -> bool: - """Validate webhook signature (GitHub, GitLab, Svix, Linear, generic HMAC-SHA256).""" + """Validate webhook signature (GitHub, GitLab, Svix, Standard Webhooks, Linear, generic HMAC-SHA256).""" headers = request.headers def _header(name: str) -> str: return headers.get(name, "") or headers.get(name.lower(), "") or headers.get(name.upper(), "") - # Svix / AgentMail: signed content is "{id}.{timestamp}.{raw_body}". + # Svix / AgentMail: signed content is "{id}.{timestamp}.{raw_body}". Standard Webhooks + # (webhook-*; GitLab signing tokens) is the same scheme under other header names. svix = [_header(name) for name in ("svix-id", "svix-timestamp", "svix-signature")] + if not any(svix): + svix = [_header(name) for name in ("webhook-id", "webhook-timestamp", "webhook-signature")] if any(svix): return _validate_svix_signature(body, secret, *svix) # Linear (any header case): hex HMAC of the body. GitHub: sha256=. GitLab: plain token. diff --git a/website/docs/user-guide/messaging/webhooks.md b/website/docs/user-guide/messaging/webhooks.md index dd1148b0fa..d9acdb8918 100644 --- a/website/docs/user-guide/messaging/webhooks.md +++ b/website/docs/user-guide/messaging/webhooks.md @@ -498,6 +498,7 @@ The adapter validates incoming webhook signatures using the appropriate method f - **GitHub**: `X-Hub-Signature-256` header — HMAC-SHA256 hex digest prefixed with `sha256=` - **GitLab**: `X-Gitlab-Token` header — plain secret string match +- **Standard Webhooks**: `webhook-id`, `webhook-timestamp`, and `webhook-signature` headers — signed content is `{id}.{timestamp}.{raw_body}` with a `v1,` signature - **Generic (V2, recommended)**: `X-Webhook-Signature-V2` + `X-Webhook-Timestamp` headers — HMAC-SHA256 hex digest of `.`. The timestamp (Unix seconds) must be within ±300 seconds of the server clock, which prevents captured requests from being replayed later. - **Generic (V1, legacy)**: `X-Webhook-Signature` header — raw HMAC-SHA256 hex digest of the body only. Still accepted for backward compatibility, but it has no replay protection (a captured request replays indefinitely); the gateway logs a deprecation warning once per route. Switch senders to V2. @@ -529,7 +530,7 @@ Requests exceeding the limit receive a `429 Too Many Requests` response. ### Idempotency -Delivery IDs (from `X-GitHub-Delivery`, `X-Request-ID`, or a timestamp fallback) are cached for **1 hour**. Duplicate deliveries (e.g. webhook retries) are silently skipped with a `200` response, preventing duplicate agent runs. +Delivery IDs (from `X-GitHub-Delivery`, `svix-id`, `webhook-id`, `X-Request-ID`, or a timestamp fallback) are cached for **1 hour**. Duplicate deliveries (e.g. webhook retries) are silently skipped with a `200` response, preventing duplicate agent runs. ### Body size limits @@ -588,7 +589,7 @@ This is the same trust model that applies to everything the agent reads: web pag ### Duplicate responses -- The idempotency cache should prevent this — check that the webhook source is sending a delivery ID header (`X-GitHub-Delivery` or `X-Request-ID`) +- The idempotency cache should prevent this — check that the webhook source is sending a delivery ID header (`X-GitHub-Delivery`, `svix-id`, `webhook-id`, or `X-Request-ID`) - Delivery IDs are cached for 1 hour ### `gh` CLI errors (GitHub comment delivery) From e40192118ee1165133197a8f0712598c7fddb70a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:35:37 -0700 Subject: [PATCH 112/276] fix(webhook): only webhook-signature commits a delivery to the Standard Webhooks path GitLab (lib/gitlab/web_hooks.rb, app/services/web_hook_service.rb) sends webhook-id and webhook-timestamp on EVERY delivery and adds webhook-signature only when a signing token is configured. Selecting the HMAC path on any webhook-* header would 401 every legacy X-Gitlab-Token install the moment it upgrades to GitLab 19, so only the signature header selects the path; id and timestamp then travel with it and the validator still fails closed when either is missing. svix-* keeps its existing any-header selection. Tests trimmed to the invariant bar: one parameterized contract (whsec_ and raw secrets accept; wrong secret, tampered body and stale timestamp reject) plus the GitLab legacy-token coexistence contract. evals/webhook_auth/standard_webhooks_ab.py drives a real aiohttp WebhookAdapter on a dedicated loopback port with real signed requests for before/after evidence. Related: #47849 (HwangJohn, cherry-picked here), #92024 (earlier salvage of #47849), #102080 and #103167 (same alias fix, same target). --- evals/webhook_auth/standard_webhooks_ab.py | 99 ++++++++++++++++++++++ gateway/platforms/webhook.py | 7 +- tests/gateway/test_webhook_adapter.py | 30 +++++++ 3 files changed, 134 insertions(+), 2 deletions(-) create mode 100644 evals/webhook_auth/standard_webhooks_ab.py diff --git a/evals/webhook_auth/standard_webhooks_ab.py b/evals/webhook_auth/standard_webhooks_ab.py new file mode 100644 index 0000000000..aa73bf98ce --- /dev/null +++ b/evals/webhook_auth/standard_webhooks_ab.py @@ -0,0 +1,99 @@ +"""Live A/B probe: Standard Webhooks (webhook-id/-timestamp/-signature) against the REAL webhook adapter. + +Starts a real ``WebhookAdapter`` (aiohttp, loopback, dedicated port) with one secret-configured +route and one ``whsec_`` route, then sends real HTTP POSTs signed exactly as the Standard +Webhooks spec (https://github.com/standard-webhooks/standard-webhooks) and GitLab's signing-token +docs describe: HMAC-SHA256 over ``{id}.{timestamp}.{body}``, ``v1,``. Cases cover the +positive path plus negative paths (wrong secret, tampered body, stale timestamp, partial headers, +wrong scheme) and the regressions that must not move (svix-*, GitLab X-Gitlab-Token, dual-token). + + HERMES_HOME=$(mktemp -d) python evals/gateway_status_render/standard_webhooks_ab.py [--port 18644] +""" + +from __future__ import annotations + +import argparse +import asyncio +import base64 +import hashlib +import hmac +import json +import os +import subprocess +import sys +import time +from unittest.mock import AsyncMock + +import aiohttp + + +def _git_head() -> str: + return subprocess.run(["git", "rev-parse", "--short", "HEAD"], capture_output=True, text=True, + stdin=subprocess.DEVNULL).stdout.strip() + + +def _sign(secret: str, msg_id: str, ts: str, body: bytes) -> str: + key = base64.b64decode(secret.removeprefix("whsec_")) if secret.startswith("whsec_") else secret.encode() + digest = hmac.new(key, f"{msg_id}.{ts}.".encode() + body, hashlib.sha256).digest() + return "v1," + base64.b64encode(digest).decode() + + +def _std_headers(secret: str, body: bytes, *, msg_id: str = "msg_1", ts: str | None = None, sign_with: str | None = None): + ts = ts or str(int(time.time())) + return {"webhook-id": msg_id, "webhook-timestamp": ts, "webhook-signature": _sign(sign_with or secret, msg_id, ts, body)} + + +async def main(port: int) -> None: + sys.path.insert(0, os.getcwd()) + from gateway.config import PlatformConfig + from gateway.platforms.webhook import WebhookAdapter + + raw_secret = "gitlab-legacy-token" + whsec = "whsec_" + base64.b64encode(b"0123456789abcdef0123456789abcdef").decode() + routes = { + "raw": {"secret": raw_secret, "prompt": "raw route"}, + "std": {"secret": whsec, "prompt": "std route"}, + } + adapter = WebhookAdapter(PlatformConfig(enabled=True, extra={"host": "127.0.0.1", "port": port, "routes": routes})) + adapter.handle_message = AsyncMock() + assert await adapter.connect(), "adapter did not bind" + body = b'{"event_type":"invoice.paid","n":1}' + stale = str(int(time.time()) - 600) + cases = [ + ("std_whsec_valid", "std", _std_headers(whsec, body), 202), + ("std_raw_secret_valid", "raw", _std_headers(raw_secret, body, msg_id="msg_2"), 202), + ("std_wrong_secret", "std", _std_headers(whsec, body, msg_id="msg_3", sign_with="whsec_" + base64.b64encode(b"x" * 32).decode()), 401), + ("std_tampered_body", "std", _std_headers(whsec, b'{"event_type":"invoice.paid","n":2}', msg_id="msg_4"), 401), + ("std_stale_timestamp_replay", "std", _std_headers(whsec, body, msg_id="msg_5", ts=stale), 401), + ("std_partial_headers_fail_closed", "std", {"webhook-id": "msg_6", "webhook-timestamp": str(int(time.time()))}, 401), + ("std_wrong_scheme_v1a", "std", {**_std_headers(whsec, body, msg_id="msg_7"), "webhook-signature": "v1a," + base64.b64encode(b"\x00" * 64).decode()}, 401), + ("std_replay_same_id", "std", _std_headers(whsec, body, msg_id="msg_1"), 200), # dedupe => 200 duplicate + ("svix_still_valid", "std", {"svix-id": "svx_1", "svix-timestamp": str(int(time.time())), "svix-signature": _sign(whsec, "svx_1", str(int(time.time())), body)}, 202), + ("gitlab_token_only_still_valid", "raw", {"X-Gitlab-Token": raw_secret}, 202), + # GitLab >= 19 sends webhook-id/webhook-timestamp on EVERY delivery and webhook-signature only when a + # signing token is configured (lib/gitlab/web_hooks.rb, app/services/web_hook_service.rb): a legacy + # secret-token install must keep working. + ("gitlab_token_with_unsigned_webhook_id_ts", "raw", {"X-Gitlab-Token": raw_secret, "webhook-id": "gl_9", "webhook-timestamp": str(int(time.time()))}, 202), + ("gitlab_dual_token_signed_with_other_key", "raw", {**_std_headers(raw_secret, body, msg_id="msg_8", sign_with="some-other-signing-token"), "X-Gitlab-Token": raw_secret}, None), + ("no_auth", "std", {}, 401), + ] + results: dict = {"head": _git_head(), "cases": []} + async with aiohttp.ClientSession() as http: + for name, route, headers, expect in cases: + # svix case: recompute with one shared ts so header and signature agree + if name == "svix_still_valid": + ts = str(int(time.time())) + headers = {"svix-id": "svx_1", "svix-timestamp": ts, "svix-signature": _sign(whsec, "svx_1", ts, body)} + async with http.post(f"http://127.0.0.1:{port}/webhooks/{route}", data=body, headers={**headers, "Content-Type": "application/json"}) as resp: + text = await resp.text() + results["cases"].append({"case": name, "route": route, "status": resp.status, "expected": expect, + "ok": (expect is None or resp.status == expect), "body": text[:120]}) + await adapter.disconnect() + results["mismatches"] = [c["case"] for c in results["cases"] if not c["ok"]] + print(json.dumps(results, indent=2)) + + +if __name__ == "__main__": + ap = argparse.ArgumentParser() + ap.add_argument("--port", type=int, default=18644) + asyncio.run(main(ap.parse_args().port)) diff --git a/gateway/platforms/webhook.py b/gateway/platforms/webhook.py index 7eb63cd505..1e7ca78b1c 100644 --- a/gateway/platforms/webhook.py +++ b/gateway/platforms/webhook.py @@ -624,9 +624,12 @@ class WebhookAdapter(BasePlatformAdapter): return headers.get(name, "") or headers.get(name.lower(), "") or headers.get(name.upper(), "") # Svix / AgentMail: signed content is "{id}.{timestamp}.{raw_body}". Standard Webhooks - # (webhook-*; GitLab signing tokens) is the same scheme under other header names. + # (webhook-*; GitLab signing tokens) is the same scheme under other header names, but GitLab + # sends webhook-id/webhook-timestamp on EVERY delivery and webhook-signature only when a signing + # token is configured, so only the signature header commits to this path — a legacy + # X-Gitlab-Token install must keep validating below (#47451, #101837). svix = [_header(name) for name in ("svix-id", "svix-timestamp", "svix-signature")] - if not any(svix): + if not any(svix) and _header("webhook-signature"): svix = [_header(name) for name in ("webhook-id", "webhook-timestamp", "webhook-signature")] if any(svix): return _validate_svix_signature(body, secret, *svix) diff --git a/tests/gateway/test_webhook_adapter.py b/tests/gateway/test_webhook_adapter.py index 7ea8e8acde..7aee89d9b4 100644 --- a/tests/gateway/test_webhook_adapter.py +++ b/tests/gateway/test_webhook_adapter.py @@ -300,6 +300,36 @@ class TestValidateSignature: ) assert adapter._validate_signature(req, body, secret) is True + @pytest.mark.parametrize( + "secret, sign_with, body, received, stale, expected", + [ + ("whsec_" + base64.b64encode(b"0123456789abcdef").decode(), None, b'{"a":1}', b'{"a":1}', False, True), + ("raw-signing-secret", None, b'{"a":1}', b'{"a":1}', False, True), + ("real-secret", "attacker-secret", b'{"a":1}', b'{"a":1}', False, False), + ("real-secret", None, b'{"a":1}', b'{"a":2}', False, False), # tampered body + ("real-secret", None, b'{"a":1}', b'{"a":1}', True, False), # replayed stale timestamp + ], + ) + def test_standard_webhooks_headers_validate_like_svix(self, secret, sign_with, body, received, stale, expected): + """#47451/#101837: webhook-id/-timestamp/-signature is the same HMAC scheme as svix-*.""" + adapter = _make_adapter() + timestamp = str(int(time.time()) - (600 if stale else 0)) + sig = _svix_signature(body, sign_with or secret, "msg_std", timestamp) + req = _mock_request(headers={"webhook-id": "msg_std", "webhook-timestamp": timestamp, "webhook-signature": sig}) + assert adapter._validate_signature(req, received, secret) is expected + + def test_gitlab_secret_token_survives_unsigned_standard_webhooks_metadata(self): + """GitLab sends webhook-id/webhook-timestamp on every delivery and webhook-signature only when a + signing token is set; a legacy X-Gitlab-Token route must not be hijacked into the HMAC path.""" + adapter = _make_adapter() + req = _mock_request(headers={ + "X-Gitlab-Token": "legacy-token", "webhook-id": "gl_1", "webhook-timestamp": str(int(time.time())), + }) + assert adapter._validate_signature(req, b"{}", "legacy-token") is True + req_partial = _mock_request(headers={"webhook-id": "gl_2", "webhook-timestamp": str(int(time.time())), + "webhook-signature": "v1,AAAA"}) + assert adapter._validate_signature(req_partial, b"{}", "legacy-token") is False + # =================================================================== # Prompt rendering From a366a0bb17db5d5dd0af327aa8f1f685c2c5dc76 Mon Sep 17 00:00:00 2001 From: turingcat Date: Tue, 11 Aug 2026 23:35:45 +0800 Subject: [PATCH 113/276] fix: route deferred update notice through cprint to avoid ANSI garbling The deferred update notice ("N commits behind") was rendered via console.print which writes ANSI escapes directly to stdout. Under patch_stdout (active in the interactive CLI), the StdoutProxy strips ESC bytes, leaving visible [1;33m...[0m artifacts. Fix: render Rich markup to an ANSI string via a captured Console, then route through cprint (which uses prompt_toolkit ANSI parser) to bypass the StdoutProxy mangling. This is the same class of bug as #2262 (fixed by #2448 for agent._print_fn and display.py), but the _defer_update_notice callsite in banner.py was not covered by that fix. Fixes #83969 --- hermes_cli/banner.py | 21 ++++++++++++--- .../tools/test_startup_latency_regressions.py | 26 +++++++++++-------- 2 files changed, 32 insertions(+), 15 deletions(-) diff --git a/hermes_cli/banner.py b/hermes_cli/banner.py index 62cab7fc26..1a1419c8b0 100644 --- a/hermes_cli/banner.py +++ b/hermes_cli/banner.py @@ -491,11 +491,25 @@ def _format_update_notice(behind: int) -> str: _deferred_update_notice_started = False +def _render_markup_to_ansi(markup: str) -> str: + """Rich markup → ANSI string, for output that must go through prompt_toolkit's renderer. + + Under ``patch_stdout`` (the interactive CLI), a plain ``Console.print`` writes ESC bytes into + the StdoutProxy, which sanitizes them into visible ``?[1;33m…`` artifacts (#83969). + """ + from io import StringIO + from rich.console import Console as _Console + buf = StringIO() + _Console(file=buf, force_terminal=True, color_system="truecolor", highlight=False).print(markup) + return buf.getvalue().rstrip("\n") + + def _defer_update_notice(console: "Console", max_wait: float = 30.0) -> None: """Print the update warning once the prefetched check completes (at most once per process). Used when the banner rendered before the update prefetch finished so startup never blocks on - git/network. + git/network. The notice lands after prompt_toolkit owns the terminal, so it is routed through + ``cprint`` (prompt_toolkit's renderer prints above a running application from any thread). """ global _deferred_update_notice_started if _deferred_update_notice_started: @@ -504,7 +518,7 @@ def _defer_update_notice(console: "Console", max_wait: float = 30.0) -> None: def _wait_and_print() -> None: if _update_check_done.wait(timeout=max_wait) and _update_result: - console.print(_format_update_notice(_update_result)) + cprint(_render_markup_to_ansi(_format_update_notice(_update_result))) _daemon("update-notice", _wait_and_print) # never break the session over an update notice @@ -863,8 +877,7 @@ def build_welcome_banner( right_lines.append(f"[dim {dim}]{' · '.join(summary_parts)}[/]") # Update check — NEVER block the banner on it: the prefetch does git/network work that rarely # finishes before render, so a blocking wait adds its full timeout to every startup. If not - # ready, a daemon thread prints the same notice above the prompt when it lands - # (prompt_toolkit's patch_stdout renders late prints safely). + # ready, a daemon thread prints the same notice above the prompt when it lands. def _update_line(): behind = get_update_result(timeout=0.05) if behind is None and not _update_check_done.is_set(): diff --git a/tests/tools/test_startup_latency_regressions.py b/tests/tools/test_startup_latency_regressions.py index c9fb3701ec..77c7c41210 100644 --- a/tests/tools/test_startup_latency_regressions.py +++ b/tests/tools/test_startup_latency_regressions.py @@ -175,37 +175,41 @@ class TestBannerUpdateCheckNonBlocking: printed = [] - class _Console: - def print(self, msg, *a, **k): - printed.append(msg) + def _fake_cprint(text): + printed.append(text) done = threading.Event() with patch.object(banner, "_update_check_done", done), \ patch.object(banner, "_update_result", None), \ - patch.object(banner, "_deferred_update_notice_started", False): - banner._defer_update_notice(_Console(), max_wait=5.0) + patch.object(banner, "_deferred_update_notice_started", False), \ + patch.object(banner, "cprint", _fake_cprint): + banner._defer_update_notice(None, max_wait=5.0) banner._update_result = 3 done.set() deadline = time.time() + 5 while not printed and time.time() < deadline: time.sleep(0.02) assert printed, "deferred update notice never printed" - assert "3 commits behind" in printed[0] + # cprint receives ANSI-rendered text (ESC escapes + visible text), + # so check that the visible payload contains the expected message. + import re + visible = re.sub(r"\x1b\[[0-9;]*m", "", printed[0]) + assert "3 commits behind" in visible def test_deferred_notice_silent_when_up_to_date(self): import hermes_cli.banner as banner printed = [] - class _Console: - def print(self, msg, *a, **k): - printed.append(msg) + def _fake_cprint(text): + printed.append(text) done = threading.Event() with patch.object(banner, "_update_check_done", done), \ patch.object(banner, "_update_result", 0), \ - patch.object(banner, "_deferred_update_notice_started", False): - banner._defer_update_notice(_Console(), max_wait=2.0) + patch.object(banner, "_deferred_update_notice_started", False), \ + patch.object(banner, "cprint", _fake_cprint): + banner._defer_update_notice(None, max_wait=2.0) done.set() time.sleep(0.3) assert not printed From 3a15c39e8e076406406f2b7be0f86537134d11f3 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:37:12 -0700 Subject: [PATCH 114/276] fix(banner): deferred update notice keeps prompt_toolkit routing; drop dead console arg; PTY harness Follow-up on the cherry-picked fix: _defer_update_notice no longer takes the Rich console it can never safely use (the notice always lands after patch_stdout owns stdout), the unit test asserts the contract at prompt_toolkit's boundary (an ANSI fragment reaches print_formatted_text, with no raw ESC/markup in the visible text) instead of patching our own cprint, and evals/cli_deferred_notice.py drives the real CLI under a Linux PTY with a FIFO-gated update cache to A/B the late-notice cases (garbled on base, clean after). --- evals/cli_deferred_notice.py | 129 ++++++++++++++++++ hermes_cli/banner.py | 4 +- .../tools/test_startup_latency_regressions.py | 31 ++--- 3 files changed, 144 insertions(+), 20 deletions(-) create mode 100644 evals/cli_deferred_notice.py diff --git a/evals/cli_deferred_notice.py b/evals/cli_deferred_notice.py new file mode 100644 index 0000000000..595e51b6ad --- /dev/null +++ b/evals/cli_deferred_notice.py @@ -0,0 +1,129 @@ +"""Drive the real CLI with a FIFO update-cache result under a Linux PTY. + +Run from the checkout with the project Python. No provider requests are sent. +The FIFO releases the real cached update result only after the prompt renders. +""" +import argparse +import errno +import json +import os +from pathlib import Path +import pty +import select +import signal +import struct +import subprocess +import sys +import tempfile +import termios +import time +import fcntl + + +def run_case(root, output, name, behind, early=False, cancel=False): + with tempfile.TemporaryDirectory(prefix="hermes_test_notice_") as home: + hh = Path(home) / ".hermes" + hh.mkdir() + (hh / "config.yaml").write_text( + "model:\n default: test-model\n provider: custom\n" + " base_url: http://127.0.0.1:9/v1\n" + "display:\n interface: cli\n skip_banner: false\n" + "memory:\n provider: ''\n", encoding="utf-8") + cache = hh / ".update_check" + # The version comes from the checkout, not an invented cache identity. + from hermes_cli.banner import VERSION + payload = json.dumps({"ts": time.time(), "behind": behind, + "rev": None, "ver": VERSION}).encode() + if early: + cache.write_bytes(payload) + else: + os.mkfifo(cache) + env = {"PATH": os.environ["PATH"], "HOME": home, "HERMES_HOME": str(hh), + "PYTHONPATH": str(root), "PYTHONUNBUFFERED": "1", + "TERM": "xterm-256color", "LANG": "C.UTF-8", + "OPENAI_API_KEY": "local-not-used", "PROMPT_TOOLKIT_NO_CPR": "1"} + master, slave = pty.openpty() + fcntl.ioctl(slave, termios.TIOCSWINSZ, struct.pack("HHHH", 40, 120, 0, 0)) + bootstrap = ("import hermes_cli.main as m; import hermes_cli.banner as b; " + "print('LOADED', m.__file__, b.__file__, flush=True); m.main()") + proc = subprocess.Popen([sys.executable, "-c", bootstrap, "chat"], cwd=root, + env=env, stdin=slave, stdout=slave, stderr=slave, + start_new_session=True) + os.close(slave) + data = bytearray() + + def pump_until(predicate, timeout=30): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if predicate(bytes(data)): + return True + if select.select([master], [], [], 0.1)[0]: + try: + chunk = os.read(master, 65536) + except OSError as exc: + if exc.errno == errno.EIO: + return predicate(bytes(data)) + raise + if not chunk: + return predicate(bytes(data)) + data.extend(chunk) + return predicate(bytes(data)) + + try: + ready = pump_until(lambda b: b"\x1b[?2004h" in b) + assert ready, f"{name}: prompt did not render: {data[-1500:]!r}" + offset = len(data) + if not early and not cancel: + fd = os.open(cache, os.O_WRONLY | os.O_NONBLOCK) + os.write(fd, payload) + os.close(fd) + pump_until(lambda b: b"to update" in b[offset:] or b"update available" in b[offset:], 3) + # prompt_toolkit prints above the prompt via run_in_terminal (input detached, + # cooked mode); type only after the prompt is redrawn, like a user would. + redraw = len(data) + pump_until(lambda b: b"\x1b[?2004h" in b[redraw:], 3) + elif cancel: + os.write(master, b"\x03") + os.write(master, b"/quit\r") + exited = pump_until(lambda b: proc.poll() is not None, 60) + proc.wait(timeout=30) + text = bytes(data).decode(errors="replace") + (output / f"{name}.pty").write_bytes(data) + result = {"case": name, "ready": ready, "exited": exited, + "returncode": proc.returncode, "garbled": "?[1;33m" in text, + "notice": "commits behind" in text or "update available" in text, + "loaded_worktree": str(root / "hermes_cli/banner.py") in text, + "raw_path": str(output / f"{name}.pty")} + assert result["loaded_worktree"] and proc.returncode == 0, result + return result + finally: + (output / f"{name}.pty").write_bytes(data) + if proc.poll() is None: + os.killpg(proc.pid, signal.SIGKILL) + proc.wait(timeout=30) + os.close(master) + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--root", required=True) + parser.add_argument("--expect", choices=("clean", "garbled"), required=True) + args = parser.parse_args() + root = Path(args.root).resolve() + args.output.mkdir(parents=True, exist_ok=True) + cases = [("late-count", 3, False, False), ("late-current", 0, False, False), + ("late-unknown-count", -1, False, False), ("early-count", 3, True, False), + ("cancel-pending", 3, False, True)] + results = [run_case(root, args.output, *case) for case in cases] + (args.output / "results.json").write_text(json.dumps(results, indent=2) + "\n") + print(json.dumps(results, indent=2)) + for row in results: + expected_notice = row["case"] not in ("late-current", "cancel-pending") + assert row["notice"] == expected_notice, row + expected_garble = args.expect == "garbled" and row["case"].startswith("late-") and expected_notice + assert row["garbled"] == expected_garble, row + + +if __name__ == "__main__": + main() diff --git a/hermes_cli/banner.py b/hermes_cli/banner.py index 1a1419c8b0..c628e2d780 100644 --- a/hermes_cli/banner.py +++ b/hermes_cli/banner.py @@ -504,7 +504,7 @@ def _render_markup_to_ansi(markup: str) -> str: return buf.getvalue().rstrip("\n") -def _defer_update_notice(console: "Console", max_wait: float = 30.0) -> None: +def _defer_update_notice(max_wait: float = 30.0) -> None: """Print the update warning once the prefetched check completes (at most once per process). Used when the banner rendered before the update prefetch finished so startup never blocks on @@ -881,7 +881,7 @@ def build_welcome_banner( def _update_line(): behind = get_update_result(timeout=0.05) if behind is None and not _update_check_done.is_set(): - _defer_update_notice(console) + _defer_update_notice() elif behind is not None and behind != 0: right_lines.append(_format_update_notice(behind)) _quiet(_update_line) # Never break the banner over an update check diff --git a/tests/tools/test_startup_latency_regressions.py b/tests/tools/test_startup_latency_regressions.py index 77c7c41210..45231f94db 100644 --- a/tests/tools/test_startup_latency_regressions.py +++ b/tests/tools/test_startup_latency_regressions.py @@ -157,44 +157,39 @@ class TestBannerUpdateCheckNonBlocking: well under the old 500ms blocking wait.""" import hermes_cli.banner as banner - class _NullConsole: - def print(self, *a, **k): - pass - with patch.object(banner, "_update_check_done", threading.Event()), \ patch.object(banner, "_deferred_update_notice_started", False): start = time.perf_counter() behind = banner.get_update_result(timeout=0.05) if behind is None and not banner._update_check_done.is_set(): - banner._defer_update_notice(_NullConsole()) + banner._defer_update_notice() elapsed = time.perf_counter() - start assert elapsed < 0.3, f"banner update check blocked {elapsed:.3f}s" - def test_deferred_notice_prints_when_result_lands(self): + def test_deferred_notice_prints_through_prompt_toolkit_renderer(self): + """The late notice lands after patch_stdout owns stdout, where raw ESC bytes are + sanitized into visible ``?[1;33m`` text (#83969). It must reach prompt_toolkit as a + parsed ANSI fragment — never as a bare ``Console.print`` to stdout.""" import hermes_cli.banner as banner + from prompt_toolkit.formatted_text import ANSI, to_formatted_text printed = [] - - def _fake_cprint(text): - printed.append(text) - done = threading.Event() with patch.object(banner, "_update_check_done", done), \ patch.object(banner, "_update_result", None), \ patch.object(banner, "_deferred_update_notice_started", False), \ - patch.object(banner, "cprint", _fake_cprint): - banner._defer_update_notice(None, max_wait=5.0) + patch("prompt_toolkit.print_formatted_text", side_effect=lambda *a, **k: printed.append(a[0])): + banner._defer_update_notice(max_wait=5.0) banner._update_result = 3 done.set() deadline = time.time() + 5 while not printed and time.time() < deadline: time.sleep(0.02) - assert printed, "deferred update notice never printed" - # cprint receives ANSI-rendered text (ESC escapes + visible text), - # so check that the visible payload contains the expected message. - import re - visible = re.sub(r"\x1b\[[0-9;]*m", "", printed[0]) + assert printed, "deferred update notice never reached prompt_toolkit's renderer" + assert isinstance(printed[0], ANSI) + visible = "".join(text for _style, text, *_ in to_formatted_text(printed[0])) assert "3 commits behind" in visible + assert "\x1b" not in visible and "[bold" not in visible def test_deferred_notice_silent_when_up_to_date(self): import hermes_cli.banner as banner @@ -209,7 +204,7 @@ class TestBannerUpdateCheckNonBlocking: patch.object(banner, "_update_result", 0), \ patch.object(banner, "_deferred_update_notice_started", False), \ patch.object(banner, "cprint", _fake_cprint): - banner._defer_update_notice(None, max_wait=2.0) + banner._defer_update_notice(max_wait=2.0) done.set() time.sleep(0.3) assert not printed From 080422216f88fb68b28992ea312393c6fc6a3147 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:37:51 -0700 Subject: [PATCH 115/276] chore(contributors): map turingcat --- contributors/emails/turingcat@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/turingcat@gmail.com diff --git a/contributors/emails/turingcat@gmail.com b/contributors/emails/turingcat@gmail.com new file mode 100644 index 0000000000..226b8a2c8b --- /dev/null +++ b/contributors/emails/turingcat@gmail.com @@ -0,0 +1 @@ +turingcat From 0543fa2f1b31783d294bf6c8b8d9ec1ebe30593b Mon Sep 17 00:00:00 2001 From: Xipong <217837358+Xipong@users.noreply.github.com> Date: Sat, 22 Aug 2026 22:54:49 +0300 Subject: [PATCH 116/276] fix(anthropic): DeepSeek thinking replay contract for third-party proxies Third-party Anthropic-compatible relays serving a DeepSeek thinking model (deepseek-r/v4/pro/flash) follow the same replay contract as DeepSeek's native /anthropic endpoint: strip signed thinking blocks, preserve unsigned ones. Previously these sessions hit the generic third-party path and lost ALL thinking blocks, breaking cross-turn reasoning coherence. Detection is model-name based (vendor prefixes stripped), gated on the existing _is_third_party check so direct Anthropic traffic is untouched. --- agent/anthropic_endpoints.py | 19 ++++++++ agent/anthropic_message_convert.py | 6 ++- tests/agent/test_anthropic_adapter.py | 25 +++++++++++ .../agent/test_deepseek_anthropic_thinking.py | 45 +++++++++++++++++++ 4 files changed, 93 insertions(+), 2 deletions(-) diff --git a/agent/anthropic_endpoints.py b/agent/anthropic_endpoints.py index 82b96c5af1..237a562176 100644 --- a/agent/anthropic_endpoints.py +++ b/agent/anthropic_endpoints.py @@ -72,6 +72,25 @@ def _is_kimi_family_endpoint(base_url: str | None, model: str | None = None) -> ) + +_DEEPSEEK_THINKING_MODEL_PREFIXES = ( + "deepseek-r", "deepseek-v4", "deepseek_v4", "deepseek-pro", + "deepseek_pro", "deepseek-flash", "deepseek_flash", +) + + +def _model_name_is_deepseek_thinking(model: str | None) -> bool: + """Known DeepSeek thinking families behind an Anthropic-compatible relay. + + Strip vendor namespaces, but do not treat arbitrary DeepSeek chat/distill + names as evidence of the thinking replay contract. + """ + if not isinstance(model, str): + return False + name = model.strip().lower().rsplit("/", 1)[-1] + return bool(name) and name.startswith(_DEEPSEEK_THINKING_MODEL_PREFIXES) + + def _is_deepseek_anthropic_endpoint(base_url: str | None) -> bool: """DeepSeek's ``/anthropic`` route. In thinking mode DeepSeek requires prior-turn ``thinking`` blocks to round-trip while the generic third-party path strips them; its blocks are unsigned, diff --git a/agent/anthropic_message_convert.py b/agent/anthropic_message_convert.py index 2335345ea7..5309ba7fc9 100644 --- a/agent/anthropic_message_convert.py +++ b/agent/anthropic_message_convert.py @@ -12,7 +12,7 @@ from typing import Any, Dict, List, Optional, Tuple from agent.anthropic_endpoints import ( _is_deepseek_anthropic_endpoint, _is_kimi_family_endpoint, _is_nous_portal_endpoint, - _is_third_party_anthropic_endpoint, + _is_third_party_anthropic_endpoint, _model_name_is_deepseek_thinking, ) logger = logging.getLogger(__name__) @@ -565,7 +565,9 @@ def _manage_thinking_signatures(result: List[Dict[str, Any]], base_url: str | No """ is_third_party = _is_third_party_anthropic_endpoint(base_url) and not _is_nous_portal_endpoint(base_url) is_kimi = _is_kimi_family_endpoint(base_url, model) - is_deepseek = _is_deepseek_anthropic_endpoint(base_url) + is_deepseek = _is_deepseek_anthropic_endpoint(base_url) or ( + is_third_party and _model_name_is_deepseek_thinking(model) + ) last_assistant_idx = next((i for i in range(len(result) - 1, -1, -1) if result[i].get("role") == "assistant"), None) for idx, m in _assistant_block_lists(result): if is_kimi: diff --git a/tests/agent/test_anthropic_adapter.py b/tests/agent/test_anthropic_adapter.py index 3ded294a6c..bbc9a822e1 100644 --- a/tests/agent/test_anthropic_adapter.py +++ b/tests/agent/test_anthropic_adapter.py @@ -1225,6 +1225,31 @@ class TestThinkingBlockSignatureManagement: + def test_third_party_deepseek_preserves_unsigned_thinking(self): + messages = [ + { + "role": "assistant", + "content": "Response text.", + "reasoning_details": [ + {"type": "thinking", "thinking": "Proxy reasoning."}, + ], + }, + ] + + _, result = convert_messages_to_anthropic( + messages, + base_url="https://proxy.example.com/anthropic", + model="deepseek-ai/deepseek-v4", + ) + + assistant = next(m for m in result if m["role"] == "assistant") + thinking = [ + block + for block in assistant["content"] + if block.get("type") == "thinking" + ] + assert thinking == [{"type": "thinking", "thinking": "Proxy reasoning."}] + def test_redacted_thinking_with_data_preserved(self): """Redacted thinking with 'data' field is kept on last turn.""" messages = [ diff --git a/tests/agent/test_deepseek_anthropic_thinking.py b/tests/agent/test_deepseek_anthropic_thinking.py index 9f0dd66f00..543ad1952c 100644 --- a/tests/agent/test_deepseek_anthropic_thinking.py +++ b/tests/agent/test_deepseek_anthropic_thinking.py @@ -105,3 +105,48 @@ class TestDeepSeekAnthropicPreservesThinking: assert "cache_control" not in b +@pytest.mark.parametrize("model", [ + "deepseek-r1", "deepseek-v4", "vendor/deepseek-v4-pro", + "gateway/deepseek-ai/deepseek_flash", " DeepSeek-Pro ", "deepseek_v4_flash", +]) +def test_thinking_family_name_detection(model): + from agent.anthropic_endpoints import _model_name_is_deepseek_thinking + assert _model_name_is_deepseek_thinking(model) + + +@pytest.mark.parametrize("model", [None, "", " ", 42, "deepseek-chat", "deepseek-v3", "vendor/", "not-deepseek-v4", "qwen-thinking"]) +def test_thinking_family_name_detection_rejects_unknown_models(model): + from agent.anthropic_endpoints import _model_name_is_deepseek_thinking + assert not _model_name_is_deepseek_thinking(model) + + +@pytest.mark.parametrize("url", [None, "https://api.anthropic.com", "https://inference-api.nousresearch.com/anthropic"]) +def test_deepseek_model_name_does_not_override_native_signature_contract(url): + from agent.anthropic_message_convert import _manage_thinking_signatures + block = {"type": "thinking", "thinking": "signed native reasoning", "signature": "sig"} + messages = [{"role": "assistant", "content": [dict(block), {"type": "text", "text": "answer"}]}] + _manage_thinking_signatures(messages, url, "deepseek-v4") + assert messages[0]["content"][0] == block + + +def test_deepseek_proxy_keeps_unsigned_thinking_in_older_tool_turns_only(): + import copy + from agent.anthropic_message_convert import convert_messages_to_anthropic + history = [ + {"role": "user", "content": "inspect"}, + {"role": "assistant", "content": "checking", "reasoning_details": [ + {"type": "thinking", "thinking": "unsigned", "cache_control": {"type": "ephemeral"}}, + {"type": "thinking", "thinking": "foreign signed", "signature": "sig"}, + {"type": "redacted_thinking", "data": "redacted-signature"}, + ], "tool_calls": [{"id": "call_1", "type": "function", "function": {"name": "inspect", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "call_1", "content": "ok"}, + {"role": "assistant", "content": "done"}, + ] + snapshot = copy.deepcopy(history) + _, result = convert_messages_to_anthropic(history, base_url="https://proxy.example/anthropic", model="vendor/deepseek-v4") + assistant = next(m for m in result if m["role"] == "assistant") + assert [b for b in assistant["content"] if b.get("type") in {"thinking", "redacted_thinking"}] == [ + {"type": "thinking", "thinking": "unsigned"}, + ] + assert history == snapshot + From ebb428aed54d0d021f1d014eb7ac489a61b8033d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:04:39 -0700 Subject: [PATCH 117/276] test(anthropic): trim DeepSeek proxy replay tests to two invariants + local wire probe Fold the adapter-level unsigned-preservation case and the name-detector matrices into one parametrized invariant (thinking family keeps unsigned only; non-thinking DeepSeek and unrelated models still take the generic third-party strip) plus the native-endpoint immutability guard. Add evals/anthropic_proxy_thinking_replay.py, a real-AIAgent probe against a synthetic local Anthropic-compatible relay used for the before/after wire check. --- evals/anthropic_proxy_thinking_replay.py | 103 ++++++++++++++++++ tests/agent/test_anthropic_adapter.py | 25 ----- .../agent/test_deepseek_anthropic_thinking.py | 30 ++--- 3 files changed, 112 insertions(+), 46 deletions(-) create mode 100644 evals/anthropic_proxy_thinking_replay.py diff --git a/evals/anthropic_proxy_thinking_replay.py b/evals/anthropic_proxy_thinking_replay.py new file mode 100644 index 0000000000..4072e4b04a --- /dev/null +++ b/evals/anthropic_proxy_thinking_replay.py @@ -0,0 +1,103 @@ +"""Local wire-contract probe: which thinking blocks a real AIAgent replays to an +Anthropic-compatible relay. Synthetic SSE fixtures; NOT live-provider proof. + +Run with the repo interpreter in an isolated HOME/HERMES_HOME: + python evals/anthropic_proxy_thinking_replay.py +""" +import json +import os +import sys +import tempfile +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) + +HISTORY = [ + {"role": "user", "content": "inspect the repo"}, + {"role": "assistant", "content": "checking", "reasoning_details": [ + {"type": "thinking", "thinking": "unsigned proxy reasoning"}, + {"type": "thinking", "thinking": "foreign signed", "signature": "sig-from-elsewhere"}, + ], "tool_calls": [{"id": "call_1", "type": "function", + "function": {"name": "terminal", "arguments": "{\"command\": \"ls\"}"}}]}, + {"role": "tool", "tool_call_id": "call_1", "content": "README.md"}, + {"role": "assistant", "content": "done"}, +] + + +def _sse(events): + return "".join(f"event: {e['type']}\ndata: {json.dumps(e)}\n\n" for e in events).encode() + + +def scenario(model, path="/anthropic"): + import run_agent + requests = [] + + class Handler(BaseHTTPRequestHandler): + def log_message(self, *_a): + pass + + def do_POST(self): + payload = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + if self.path.endswith("/messages"): # skip the Ollama /api/show capability probe + requests.append(payload) + msg = {"id": "msg_local", "type": "message", "role": "assistant", "model": payload.get("model"), + "content": [], "stop_reason": None, "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 0}} + events = [ + {"type": "message_start", "message": msg}, + {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}, + {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "local fixture reply"}}, + {"type": "content_block_stop", "index": 0}, + {"type": "message_delta", "delta": {"stop_reason": "end_turn", "stop_sequence": None}, + "usage": {"output_tokens": 3}}, + {"type": "message_stop"}, + ] + raw = _sse(events) + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + threading.Thread(target=server.serve_forever, daemon=True).start() + agent = run_agent.AIAgent( + model=model, provider="custom", base_url=f"http://127.0.0.1:{server.server_port}{path}", + api_key="local-fixture-key", enabled_toolsets=["terminal"], max_iterations=1, quiet_mode=True, + reasoning_config={"enabled": True, "effort": "high"}, skip_context_files=True, skip_memory=True, + skip_background_review=True, save_trajectories=False, session_id=f"proxy-thinking-{model.replace('/', '-')}", + ) + agent._api_max_retries = 1 + try: + result = agent.run_conversation("reply briefly", system_message="Local wire probe.", + conversation_history=[dict(m) for m in HISTORY]) + finally: + agent.close() + server.shutdown() + server.server_close() + wire = requests[0] if requests else {} + assistant_turns = [m for m in wire.get("messages", []) if m.get("role") == "assistant"] + thinking = [ + {k: b.get(k) for k in ("type", "thinking", "signature") if k in b} + for m in assistant_turns for b in (m.get("content") if isinstance(m.get("content"), list) else []) + if b.get("type") in ("thinking", "redacted_thinking") + ] + return {"model": model, "api_mode": agent.api_mode, "completed": result.get("completed"), + "requests": len(requests), "wire_thinking_blocks": thinking, + "thinking_param": wire.get("thinking"), "tool_count": len(wire.get("tools") or [])} + + +def main(): + out = {"surface": "real AIAgent + Anthropic SDK against a synthetic local relay; not provider proof", + "scenarios": [scenario("deepseek-ai/deepseek-v4-pro"), scenario("some-vendor/other-model")]} + Path(sys.argv[1]).write_text(json.dumps(out, indent=2) + "\n", encoding="utf-8") + print(json.dumps(out, indent=2)) + + +if __name__ == "__main__": + with tempfile.TemporaryDirectory(prefix="proxy-thinking-") as home: + os.environ.update(HOME=home, HERMES_HOME=home, HERMES_DISABLE_PLUGINS="1") + main() diff --git a/tests/agent/test_anthropic_adapter.py b/tests/agent/test_anthropic_adapter.py index bbc9a822e1..3ded294a6c 100644 --- a/tests/agent/test_anthropic_adapter.py +++ b/tests/agent/test_anthropic_adapter.py @@ -1225,31 +1225,6 @@ class TestThinkingBlockSignatureManagement: - def test_third_party_deepseek_preserves_unsigned_thinking(self): - messages = [ - { - "role": "assistant", - "content": "Response text.", - "reasoning_details": [ - {"type": "thinking", "thinking": "Proxy reasoning."}, - ], - }, - ] - - _, result = convert_messages_to_anthropic( - messages, - base_url="https://proxy.example.com/anthropic", - model="deepseek-ai/deepseek-v4", - ) - - assistant = next(m for m in result if m["role"] == "assistant") - thinking = [ - block - for block in assistant["content"] - if block.get("type") == "thinking" - ] - assert thinking == [{"type": "thinking", "thinking": "Proxy reasoning."}] - def test_redacted_thinking_with_data_preserved(self): """Redacted thinking with 'data' field is kept on last turn.""" messages = [ diff --git a/tests/agent/test_deepseek_anthropic_thinking.py b/tests/agent/test_deepseek_anthropic_thinking.py index 543ad1952c..48cce7d8bd 100644 --- a/tests/agent/test_deepseek_anthropic_thinking.py +++ b/tests/agent/test_deepseek_anthropic_thinking.py @@ -105,21 +105,6 @@ class TestDeepSeekAnthropicPreservesThinking: assert "cache_control" not in b -@pytest.mark.parametrize("model", [ - "deepseek-r1", "deepseek-v4", "vendor/deepseek-v4-pro", - "gateway/deepseek-ai/deepseek_flash", " DeepSeek-Pro ", "deepseek_v4_flash", -]) -def test_thinking_family_name_detection(model): - from agent.anthropic_endpoints import _model_name_is_deepseek_thinking - assert _model_name_is_deepseek_thinking(model) - - -@pytest.mark.parametrize("model", [None, "", " ", 42, "deepseek-chat", "deepseek-v3", "vendor/", "not-deepseek-v4", "qwen-thinking"]) -def test_thinking_family_name_detection_rejects_unknown_models(model): - from agent.anthropic_endpoints import _model_name_is_deepseek_thinking - assert not _model_name_is_deepseek_thinking(model) - - @pytest.mark.parametrize("url", [None, "https://api.anthropic.com", "https://inference-api.nousresearch.com/anthropic"]) def test_deepseek_model_name_does_not_override_native_signature_contract(url): from agent.anthropic_message_convert import _manage_thinking_signatures @@ -129,7 +114,13 @@ def test_deepseek_model_name_does_not_override_native_signature_contract(url): assert messages[0]["content"][0] == block -def test_deepseek_proxy_keeps_unsigned_thinking_in_older_tool_turns_only(): +@pytest.mark.parametrize(("model", "kept"), [ + ("vendor/deepseek-v4", [{"type": "thinking", "thinking": "unsigned"}]), # thinking family: keep unsigned only + (" DeepSeek-Pro ", [{"type": "thinking", "thinking": "unsigned"}]), + ("deepseek-chat", []), # non-thinking DeepSeek and unrelated models: generic third-party strip + ("vendor/other-model", []), +]) +def test_deepseek_proxy_keeps_unsigned_thinking_in_older_tool_turns_only(model, kept): import copy from agent.anthropic_message_convert import convert_messages_to_anthropic history = [ @@ -143,10 +134,7 @@ def test_deepseek_proxy_keeps_unsigned_thinking_in_older_tool_turns_only(): {"role": "assistant", "content": "done"}, ] snapshot = copy.deepcopy(history) - _, result = convert_messages_to_anthropic(history, base_url="https://proxy.example/anthropic", model="vendor/deepseek-v4") + _, result = convert_messages_to_anthropic(history, base_url="https://proxy.example/anthropic", model=model) assistant = next(m for m in result if m["role"] == "assistant") - assert [b for b in assistant["content"] if b.get("type") in {"thinking", "redacted_thinking"}] == [ - {"type": "thinking", "thinking": "unsigned"}, - ] + assert [b for b in assistant["content"] if b.get("type") in {"thinking", "redacted_thinking"}] == kept assert history == snapshot - From 67807e64a66044db9e0a641d98c68a35c1760589 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:30:17 -0700 Subject: [PATCH 118/276] chore(contributors): map Xipong noreply email --- contributors/emails/217837358+Xipong@users.noreply.github.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/217837358+Xipong@users.noreply.github.com diff --git a/contributors/emails/217837358+Xipong@users.noreply.github.com b/contributors/emails/217837358+Xipong@users.noreply.github.com new file mode 100644 index 0000000000..4e1370864b --- /dev/null +++ b/contributors/emails/217837358+Xipong@users.noreply.github.com @@ -0,0 +1,2 @@ +Xipong +# PR #92488 / #76370 salvage From e0592b6d43322c58cd59b5d7e6bf3bce53ae0faf Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:54:32 -0700 Subject: [PATCH 119/276] fix(codex): masked "invalid_prompt: Request blocked." replay rejection reaches the replay-strip recovery MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The ChatGPT Codex backend answers a rejected encrypted-reasoning replay with the same bare 400 {code: invalid_prompt, message: "Request blocked."} it uses for genuine blocks. classify_api_error bucketed it as format_error, so the one-shot repair that already exists for invalid_encrypted_content (disable replay, strip cached codex_reasoning_items, retry once) never ran and the turn aborted. Classify exactly that envelope from provider openai-codex — SDK 400 body, SSE ``error`` frame, or the ``response.failed`` text — as invalid_encrypted_content while keeping format_error's abort-and-fallback hints, so the only behavioural delta is turn_recovery's replay strip, which still requires cached reasoning items. A block with nothing to strip aborts exactly as before; other providers, other messages/codes and the #18028 safety refusals are untouched. evals/codex_masked_replay_review.py drives a real AIAgent (codex_responses) against a synthetic local Responses server for the before/after check. Refs #92353, #92357, #94077. --- agent/error_classifier.py | 23 +++++ evals/codex_masked_replay_review.py | 138 +++++++++++++++++++++++++++ tests/agent/test_error_classifier.py | 25 +++++ 3 files changed, 186 insertions(+) create mode 100644 evals/codex_masked_replay_review.py diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 7bf3b8716f..177c5bbd84 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -486,6 +486,13 @@ def _provider_special_cases(c: _Ctx) -> Optional[Verdict]: # to format_error and a status-less block isn't left retryable (#18028). if any(p in msg for p in _CONTENT_POLICY_BLOCKED_PATTERNS): return _V_CONTENT_BLOCKED + # ChatGPT Codex masks a rejected encrypted-reasoning replay behind the same bare + # ``invalid_prompt: Request blocked.`` it uses for real blocks (#92353). Exact envelope + # + provider only. The verdict keeps format_error's abort-and-fallback hints; the one + # extra thing it buys is turn_recovery's replay strip, which still requires cached + # ``codex_reasoning_items`` — a genuine block with nothing to strip behaves as before. + if _is_codex_masked_replay_rejection(c): + return _v(_R.invalid_encrypted_content, **_ABORT_FALLBACK) # Anthropic thinking-block 400s (signature mismatch after transcript # mutation). Not gated on provider — OpenRouter proxies Anthropic errors. if status == 400 and "thinking" in msg and any(p in msg for p in _THINKING_MUTATION_WORDS): @@ -777,6 +784,22 @@ def _is_server_injected_param_rejection(error_msg: str, provider: str) -> bool: return False +_CODEX_MASKED_REPLAY_MESSAGE = "request blocked." + + +def _is_codex_masked_replay_rejection(c: "_Ctx") -> bool: + """HTTP 400 / status-less ``{code: invalid_prompt, message: "Request blocked."}`` from + ``openai-codex`` — as an SDK error body, a Responses ``error`` SSE frame, or the + ``response.failed`` text ``"invalid_prompt: Request blocked."``.""" + if c.provider_slug != "openai-codex" or c.status_code not in (None, 400): + return False + # The OpenAI SDK unwraps ``body["error"]`` on status errors; stream frames keep the envelope. + body_msg = next((str(m).strip().lower() for m in _body_message_candidates(c.body or {}) if m), "") + return (c.code == "invalid_prompt" and body_msg == _CODEX_MASKED_REPLAY_MESSAGE) or ( + c.msg.strip() == f"invalid_prompt: {_CODEX_MASKED_REPLAY_MESSAGE}" + ) + + def _error_obj(body: Any) -> dict: """``body["error"]`` when it is a dict, else ``{}``.""" err = body.get("error") if isinstance(body, dict) else None diff --git a/evals/codex_masked_replay_review.py b/evals/codex_masked_replay_review.py new file mode 100644 index 0000000000..971dacc19f --- /dev/null +++ b/evals/codex_masked_replay_review.py @@ -0,0 +1,138 @@ +"""Local HTTP contract review of masked-error recovery; NOT live-provider proof. + +Run in an isolated HOME/HERMES_HOME under the campaign test lock. Responses are +explicit synthetic fixtures; the client, transport, and AIAgent loop are real. +""" +import copy +import hashlib +import json +import os +from pathlib import Path +import sys +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(ROOT)) +import run_agent +from agent import error_classifier +from openai import APIError +import httpx + +ITEM = {"type": "reasoning", "id": "rs_review", "encrypted_content": "signed-control-opaque-do-not-alter", "summary": []} +ERROR = {"message": "Request blocked.", "type": "invalid_request_error", "param": None, "code": "invalid_prompt"} + + +def digest(value): + return hashlib.sha256(json.dumps(value, sort_keys=True).encode()).hexdigest() + + +def scenario(name, provider="openai-codex", replay=True): + requests = [] + + class Handler(BaseHTTPRequestHandler): + def log_message(self, *_args): + pass + + def do_POST(self): + payload = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + if "input" not in payload: + raw = json.dumps({"id": "chat-local", "object": "chat.completion", "created": 1, + "model": "gpt-5-codex", "choices": [{"index": 0, "finish_reason": "stop", + "message": {"role": "assistant", "content": "local auxiliary fixture"}}]}).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + self.wfile.write(raw) + return + requests.append(payload) + has_replay = any(i.get("type") == "reasoning" for i in payload["input"]) + error = copy.deepcopy(ERROR) + if name == "explicit_encrypted": + error.update(code="invalid_encrypted_content", message="The encrypted content could not be verified.") + reject = name in {"unrelated_block", "failed_frame"} or (name not in {"success", "failed_frame"} and has_replay) + if reject and name == "failed_frame": + # The shape openai/codex issues show for a blocked turn: response.in_progress, then a + # ``response.failed`` terminal frame carrying the error (no HTTP status). + failed = {"id": "resp_review", "object": "response", "created_at": 1, "status": "failed", + "model": "gpt-5-codex", "output": [], "error": error} + events = [{"type": "response.created", "response": dict(failed, status="in_progress", error=None), "sequence_number": 0}, + {"type": "response.in_progress", "response": dict(failed, status="in_progress", error=None), "sequence_number": 1}, + {"type": "response.failed", "response": failed, "sequence_number": 2}] + raw = "".join("data: " + json.dumps(e) + "\n\n" for e in events).encode() + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + elif reject: + raw = json.dumps({"error": error}).encode() + self.send_response(400) + self.send_header("Content-Type", "application/json") + else: + response = {"id": "resp_review", "object": "response", "created_at": 1, + "status": "completed", "model": "gpt-5-codex", "output": [ + {"id": "msg_review", "type": "message", "role": "assistant", "status": "completed", + "content": [{"type": "output_text", "text": "local fixture success", "annotations": []}]}], + "usage": {"input_tokens": 10, "output_tokens": 3, "total_tokens": 13}} + created = dict(response, status="in_progress", output=[]) + events = [{"type": "response.created", "response": created, "sequence_number": 0}, + {"type": "response.output_item.done", "item": response["output"][0], "output_index": 0, "sequence_number": 1}, + {"type": "response.completed", "response": response, "sequence_number": 2}] + raw = "".join("data: " + json.dumps(e) + "\n\n" for e in events).encode() + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Content-Length", str(len(raw))) + self.send_header("x-request-id", f"local-{name}-{len(requests)}") + self.end_headers() + self.wfile.write(raw) + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + agent = run_agent.AIAgent( + model="gpt-5-codex", provider=provider, api_mode="codex_responses", + base_url=f"http://127.0.0.1:{server.server_port}/v1", api_key="local-fixture-key", + enabled_toolsets=["terminal"], max_iterations=2, quiet_mode=True, + reasoning_config={"effort": "high"}, skip_context_files=True, + skip_memory=True, skip_background_review=True, save_trajectories=False, + session_id=f"local-review-{name}-{provider}-{replay}", + ) + agent._api_max_retries = 3 # show whether a deterministic rejection burns identical retries + history = [{"role": "user", "content": "hello"}, {"role": "assistant", "content": "hello"}] + if replay: + history[1]["codex_reasoning_items"] = [copy.deepcopy(ITEM)] + try: + result = agent.run_conversation("reply briefly", system_message="Local contract review.", conversation_history=history) + reasoning_counts = [sum(i.get("type") == "reasoning" for i in p["input"]) for p in requests] + stable = {k: len({digest(p.get(k)) for p in requests}) <= 1 + for k in ["reasoning", "instructions", "tools"]} + signed_before = [i for i in requests[0]["input"] if i.get("type") == "reasoning"] if requests else [] + return {"name": name, "provider": provider, "replay": replay, "completed": result.get("completed"), + "requests": len(requests), "reasoning_counts": reasoning_counts, + "stable_retry_fields": stable, "reasoning_parameters": [p.get("reasoning") for p in requests], + "tool_count": len(requests[0].get("tools", [])) if requests else 0, + "signed_payload_unchanged": [i.get("encrypted_content") for i in signed_before] == ([ITEM["encrypted_content"]] if replay else []), + "replay_enabled_after": agent._codex_reasoning_replay_enabled, + "canonical_replay_after": any(m.get("codex_reasoning_items") for m in result["messages"])} + finally: + agent.close() + server.shutdown() + server.server_close() + thread.join() + + +def main(): + print(json.dumps({"run_agent": run_agent.__file__, "classifier": error_classifier.__file__, + "home": os.environ["HERMES_HOME"]})) + results = [scenario("success"), scenario("masked_replay"), scenario("explicit_encrypted"), + scenario("unrelated_block"), scenario("unrelated_block", replay=False), + scenario("masked_replay", provider="custom"), scenario("failed_frame"), scenario("failed_frame", replay=False)] + e = APIError("Request blocked.", request=httpx.Request("POST", "http://127.0.0.1"), body=ERROR) + output = {"surface": "real AIAgent and SDK against synthetic local HTTP/SSE server; not provider proof", + "statusless_reason": error_classifier.classify_api_error(e, provider="openai-codex").reason.value, + "scenarios": results} + Path(sys.argv[1]).write_text(json.dumps(output, indent=2) + "\n", encoding="utf-8") + print(json.dumps(output, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index 8b29100a48..b6156a9dc8 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -691,6 +691,31 @@ class TestClassifyApiError: assert result.retryable is True assert result.should_fallback is False + # ── Codex masked encrypted-reasoning replay rejection (#92353) ── + + _CODEX_MASKED = {"message": "Request blocked.", "type": "invalid_request_error", "param": None, "code": "invalid_prompt"} + + @pytest.mark.parametrize("error", [ + MockAPIError("Error code: 400 - Request blocked.", status_code=400, body=_CODEX_MASKED), # SDK unwraps body["error"] + MockAPIError("Request blocked.", status_code=None, body={"error": _CODEX_MASKED}), # SSE ``error`` frame + RuntimeError("invalid_prompt: Request blocked."), # ``response.failed`` terminal frame + ], ids=["http400", "sse-frame", "response-failed"]) + def test_codex_masked_replay_rejection_reaches_replay_strip(self, error): + result = classify_api_error(error, provider="openai-codex", model="gpt-5.5") + assert result.reason == FailoverReason.invalid_encrypted_content + assert result.retryable is False and result.should_fallback is True # format_error's terminal hints kept + + @pytest.mark.parametrize(("provider", "body", "expected"), [ + ("custom", _CODEX_MASKED, FailoverReason.format_error), # same envelope, other provider + ("openai-codex", {**_CODEX_MASKED, "message": "Invalid prompt: too long."}, FailoverReason.format_error), + ("openai-codex", {**_CODEX_MASKED, "code": "server_error"}, FailoverReason.format_error), + ("openai-codex", {**_CODEX_MASKED, "message": "Request blocked. Your request was flagged by our safety system."}, + FailoverReason.content_policy_blocked), # #18028 refusal still wins + ], ids=["other-provider", "other-message", "other-code", "safety-refusal"]) + def test_codex_masked_replay_rejection_stays_narrow(self, provider, body, expected): + e = MockAPIError("Error code: 400 - " + body["message"], status_code=400, body=body) + assert classify_api_error(e, provider=provider, model="gpt-5.5").reason == expected + # ── Reasoning-mandatory route rejecting a disable ── def test_reasoning_mandatory_400_is_retryable_not_format_error(self): From 6efca62c20e97f6503298d0f57d6d7968ce067ab Mon Sep 17 00:00:00 2001 From: Glucksberg <80581902+Glucksberg@users.noreply.github.com> Date: Sat, 5 Sep 2026 13:20:48 -0300 Subject: [PATCH 120/276] [verified] fix(gateway): preserve complete resolved reply context --- gateway/run_inbound.py | 6 ++++-- tests/gateway/test_reply_to_injection.py | 26 ++++++++++++++++++++++++ 2 files changed, 30 insertions(+), 2 deletions(-) diff --git a/gateway/run_inbound.py b/gateway/run_inbound.py index 989ccf794e..b1e223a632 100644 --- a/gateway/run_inbound.py +++ b/gateway/run_inbound.py @@ -1497,9 +1497,11 @@ class GatewayInboundMixin: if getattr(event, "reply_to_text", None) and event.reply_to_message_id: # Always inject the reply-to pointer even when the quoted text is already in history: # it's disambiguation (*which* prior message), not deduplication. - reply_snippet = event.reply_to_text[:500] + # Adapters resolve the original message (or the user's native partial quote). + # A preview here silently loses later list items and code; keep that context intact. + reply_text = event.reply_to_text _who = " your previous message" if getattr(event, "reply_to_is_own_message", False) else "" - message_text = f'[Replying to{_who}: "{reply_snippet}"]\n\n{message_text}' + message_text = f'[Replying to{_who}: "{reply_text}"]\n\n{message_text}' return message_text async def _inbound_model_context_length(self, source: SessionSource, session_key: str) -> int: diff --git a/tests/gateway/test_reply_to_injection.py b/tests/gateway/test_reply_to_injection.py index d92e6a53e0..b536c8a24d 100644 --- a/tests/gateway/test_reply_to_injection.py +++ b/tests/gateway/test_reply_to_injection.py @@ -60,6 +60,32 @@ async def test_reply_prefix_injected_when_text_absent_from_history(): assert result.endswith("What's the best time to go?") +@pytest.mark.asyncio +async def test_telegram_long_reply_reaches_prompt_without_losing_later_items(): + """The native reply already has the full message; preparation must not trim it.""" + from gateway.platforms.base import MessageType + from tests.gateway.test_telegram_reply_quote import _make_adapter, _make_message + + quoted = "\n".join( + f"{index}. {company}: " + "Evidence from the supplied list. " * 12 + for index, company in enumerate( + ["GoCar", "Urban Drive", "DubCar", "GRPS", "Halucar"], 1 + ) + ) + event = _make_adapter()._build_message_event( + _make_message(text="Review all five companies.", reply_to_text=quoted), + MessageType.TEXT, + ) + history = [{"role": "user", "content": "Previous request"}] + result = await _make_runner()._prepare_inbound_message_text( + event=event, source=event.source, history=history, + ) + assert result is not None + assert quoted in result + assert result.endswith("Review all five companies.") + assert history == [{"role": "user", "content": "Previous request"}] + + @pytest.mark.asyncio async def test_reply_prefix_still_injected_when_text_in_history(): """Regression test: the pointer must survive even when the quoted text From 43ddf50b04ccb14c199c538bfee108591f1a64c7 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:41:18 -0700 Subject: [PATCH 121/276] fix(gateway): expand @ references before the reply pointer so quoted text stays literal Follow-up to the truncation removal: with the full reply text in the prepared message, a `@file:`/`@url:` reference inside the *quoted* message reached the context-reference expander and read a local file on the replier's behalf (the old 500-char slice hid this for long quotes; short quotes were already exposed on main). Expansion now runs on the new message text only; the reply pointer is prepended afterwards. --- gateway/run_inbound.py | 9 +++++--- tests/gateway/test_reply_to_injection.py | 27 ++++++++++++++++++++++++ 2 files changed, 33 insertions(+), 3 deletions(-) diff --git a/gateway/run_inbound.py b/gateway/run_inbound.py index b1e223a632..1208221fc5 100644 --- a/gateway/run_inbound.py +++ b/gateway/run_inbound.py @@ -1618,10 +1618,13 @@ class GatewayInboundMixin: message_text = await self._enrich_inbound_voice(event, source, message_text, audio_paths) message_text = self._prepend_inbound_media_file_notes(message_text, audio_file_paths, video_paths) message_text = self._prepend_inbound_document_notes(event, message_text) - message_text = self._prepend_inbound_reply_context(event, source, message_text) if "@" in message_text: - return await self._expand_inbound_context_references(source, session_key, message_text) - return message_text + message_text = await self._expand_inbound_context_references(source, session_key, message_text) + if message_text is None: + return None + # After expansion: the quoted reply is someone else's text and stays literal — an + # ``@file:`` inside it must never read a local file on the replier's behalf. + return self._prepend_inbound_reply_context(event, source, message_text) async def _prepare_profile_scoped_inbound_message_text( self, *, event: MessageEvent, source: SessionSource, history: List[Dict[str, Any]], diff --git a/tests/gateway/test_reply_to_injection.py b/tests/gateway/test_reply_to_injection.py index b536c8a24d..bc6f9fe7ab 100644 --- a/tests/gateway/test_reply_to_injection.py +++ b/tests/gateway/test_reply_to_injection.py @@ -86,6 +86,33 @@ async def test_telegram_long_reply_reaches_prompt_without_losing_later_items(): assert history == [{"role": "user", "content": "Previous request"}] +@pytest.mark.asyncio +async def test_quoted_reply_references_stay_literal_while_typed_ones_expand(tmp_path, monkeypatch): + """The replied-to author's ``@file:`` is quoted text, not the replier's request: no local read. + The same reference typed in the new message still expands (positive control).""" + import threading + + payload = tmp_path / "notes.txt" + payload.write_text("LOCAL-FILE-MARKER", encoding="utf-8") + monkeypatch.setenv("TERMINAL_CWD", str(tmp_path)) + runner = _make_runner() + runner._session_model_overrides, runner._last_resolved_model = {}, {} + runner._agent_cache, runner._agent_cache_lock = {}, threading.Lock() + runner._resolve_session_agent_runtime = lambda **kw: ("openai/gpt-4.1-mini", {"base_url": None, "api_key": ""}) + source = _source() + + quoted = ("x " * 300) + f"\nsee @file:{payload.name} for details" + quoted_ref = MessageEvent(text="what does this say?", source=source, reply_to_message_id="7", reply_to_text=quoted) + result = await runner._prepare_inbound_message_text(event=quoted_ref, source=source, history=[]) + assert quoted in result + assert "LOCAL-FILE-MARKER" not in result + + typed_ref = MessageEvent(text=f"read @file:{payload.name}", source=source, reply_to_message_id="7", reply_to_text="short") + result = await runner._prepare_inbound_message_text(event=typed_ref, source=source, history=[]) + assert result.startswith('[Replying to: "short"]') + assert "LOCAL-FILE-MARKER" in result + + @pytest.mark.asyncio async def test_reply_prefix_still_injected_when_text_in_history(): """Regression test: the pointer must survive even when the quoted text From bc233501aa115dcc80e600c4aebd2964ec568d1e Mon Sep 17 00:00:00 2001 From: Glucksberg <80581902+Glucksberg@users.noreply.github.com> Date: Fri, 4 Sep 2026 22:18:13 -0300 Subject: [PATCH 122/276] fix(telegram): preserve group mentions and expose per-turn addressing --- plugins/platforms/telegram/adapter.py | 13 ++- .../platforms/telegram/telegram_context.py | 27 +++++ .../gateway/test_telegram_mention_context.py | 101 ++++++++++++++++++ website/docs/user-guide/messaging/telegram.md | 2 + 4 files changed, 138 insertions(+), 5 deletions(-) create mode 100644 plugins/platforms/telegram/telegram_context.py create mode 100644 tests/gateway/test_telegram_mention_context.py diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 73afb13b0f..555ba218cf 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -5657,9 +5657,10 @@ class TelegramAdapter(BasePlatformAdapter): return False async def _build_triggered_event(self, msg, update, msg_type: MessageType) -> MessageEvent: - """Event for an addressed text/command: trigger text cleaned, replied-to media cached, attribution applied.""" + """Keep group conversation addressing; normalize command triggers for dispatch.""" event = self._build_message_event(msg, msg_type, update_id=update.update_id) - event.text = self._clean_bot_trigger_text(event.text) + if msg_type == MessageType.COMMAND or not self._is_group_chat(msg): + event.text = self._clean_bot_trigger_text(event.text) await self._cache_replied_media(msg, event) return self._apply_telegram_group_observe_attribution(event) @@ -5978,13 +5979,13 @@ class TelegramAdapter(BasePlatformAdapter): if self._should_observe_unmentioned_group_message(msg): _event = self._build_message_event(msg, self._media_message_type(msg), update_id=update.update_id) if msg.caption: - _event.text = self._clean_bot_trigger_text(msg.caption) + _event.text = msg.caption await self._cache_observed_media(msg, _event) self._observe_unmentioned_group_message(msg, _event.message_type, update_id=update.update_id, event=_event) return event = self._build_message_event(msg, self._media_message_type(msg), update_id=update.update_id) if msg.caption: - event.text = self._clean_bot_trigger_text(msg.caption) + event.text = msg.caption if self._is_group_chat(msg) else self._clean_bot_trigger_text(msg.caption) # Stickers: _handle_sticker overwrites event.text with its vision description, so observe attribution must run after it. if msg.sticker: await self._handle_sticker(msg, event) @@ -6281,12 +6282,14 @@ class TelegramAdapter(BasePlatformAdapter): is_bot=bool(getattr(user, "is_bot", False)) if user else False) reply_to_id, reply_to_text = self._reply_context(message) from gateway.platforms.base import resolve_channel_prompt # per-channel/topic ephemeral prompt + from plugins.platforms.telegram.telegram_context import group_addressing_prompt _chat_id_str = str(chat.id) + channel_prompt = resolve_channel_prompt(self.config.extra, thread_id_str or _chat_id_str, _chat_id_str if thread_id_str else None) return MessageEvent( text=message.text or "", message_type=msg_type, source=source, raw_message=message, message_id=str(message.message_id), platform_update_id=update_id, reply_to_message_id=reply_to_id, reply_to_text=reply_to_text, auto_skill=topic_skill, - channel_prompt=resolve_channel_prompt(self.config.extra, thread_id_str or _chat_id_str, _chat_id_str if thread_id_str else None), + channel_prompt=group_addressing_prompt(self, message, channel_prompt), timestamp=message.date) # -- Message reactions (processing lifecycle) -- diff --git a/plugins/platforms/telegram/telegram_context.py b/plugins/platforms/telegram/telegram_context.py new file mode 100644 index 0000000000..a4ab9ada1b --- /dev/null +++ b/plugins/platforms/telegram/telegram_context.py @@ -0,0 +1,27 @@ +"""Per-message Telegram addressing facts, outside the cached session prompt.""" + +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from telegram import Message + from plugins.platforms.telegram.adapter import TelegramAdapter + + +def group_addressing_prompt( + adapter: "TelegramAdapter", message: "Message", channel_prompt: str | None, +) -> str | None: + if not adapter._is_group_chat(message) or not getattr(adapter, "_bot", None): + return channel_prompt + username = adapter._current_bot_username() + if not username: + return channel_prompt + # Use the same entity-aware check as admission, not the cleaned text or the + # fact that a turn was dispatched (replies/wake words/open groups also pass). + mentioned = "yes" if adapter._message_mentions_bot(message) else "no" + addressing = ( + "Telegram addressing context (current message only):\n" + f"- Your Telegram bot username: @{username}\n" + f"- Current message explicitly mentions you: {mentioned}\n" + "Mentions of other bots do not by themselves ask you to relay the message." + ) + return f"{channel_prompt}\n\n{addressing}" if channel_prompt else addressing diff --git a/tests/gateway/test_telegram_mention_context.py b/tests/gateway/test_telegram_mention_context.py new file mode 100644 index 0000000000..35c36985f0 --- /dev/null +++ b/tests/gateway/test_telegram_mention_context.py @@ -0,0 +1,101 @@ +"""Addressing information must survive routing into the agent's event.""" + +import asyncio +from types import SimpleNamespace +from unittest.mock import AsyncMock + +import pytest + +from gateway.platforms.base import MessageType +from tests.gateway.test_telegram_group_gating import ( + _dm_message, _group_message, _group_voice_message, _make_adapter, + _mention_entities, +) + + +@pytest.mark.parametrize("media", [False, True]) +@pytest.mark.parametrize("observe", [False, True]) +def test_multi_bot_addressing_survives_real_handlers(media, observe): + async def run(): + text = "@research_bot , @ops_bot are you both listening?" + for username in ("research_bot", "ops_bot", "unrelated_bot"): + adapter = _make_adapter( + bot_username=username, require_mention=True, + exclusive_bot_mentions=True, observe_unmentioned_group_messages=observe, + allowed_chats=["-100"], group_allowed_chats=["-100"], + ) + adapter.config.extra["channel_prompts"] = {"-100": "Keep answers concise."} + events = [] + adapter._enqueue_text_event = events.append + adapter.handle_message = AsyncMock(side_effect=events.append) + adapter._ensure_forum_commands = AsyncMock() + adapter._cache_inbound_av = AsyncMock(return_value=False) + entities = _mention_entities(text, ["@research_bot", "@ops_bot"]) + if media: + msg = _group_voice_message(caption=text) + msg.caption_entities = entities + handler = adapter._handle_media_message + else: + msg = _group_message(text, entities=entities) + handler = adapter._handle_text_message + update = SimpleNamespace(update_id=1001, message=msg, effective_message=None) + + await handler(update, SimpleNamespace()) + + if username == "unrelated_bot": + assert not events + continue + assert len(events) == 1 + event = events[0] + assert text in event.text # Preserve both recipients and their positions. + assert f"Your Telegram bot username: @{username}" in event.channel_prompt + assert "Current message explicitly mentions you: yes" in event.channel_prompt + assert "Keep answers concise." in event.channel_prompt + assert ("observed Telegram group context" in event.channel_prompt) == observe + assert event.source.user_id == (None if observe else "111") + + asyncio.run(run()) + + +@pytest.mark.parametrize("trigger", ["mention", "text_mention", "reply", "wake_word", "open", "code", "command", "dm"]) +def test_addressing_context_reports_original_entities_without_inventing_mentions(trigger): + async def run(): + adapter = _make_adapter(require_mention=trigger != "open", mention_patterns=["^wake\\b"]) + adapter._ensure_forum_commands = AsyncMock() + events = [] + adapter._enqueue_text_event = events.append + adapter.handle_message = AsyncMock(side_effect=events.append) + msg_type = MessageType.TEXT + if trigger == "dm": + msg = _dm_message("hello") + elif trigger == "command": + msg_type = MessageType.COMMAND + msg = _group_message("/new@hermes_bot", entities=[SimpleNamespace(type="bot_command", offset=0, length=15)]) + elif trigger == "text_mention": + msg = _group_message("Hermes hello", entities=[SimpleNamespace(type="text_mention", offset=0, length=6, user=SimpleNamespace(id=999))]) + elif trigger == "mention": + text = "😀 @hermes_bot hello" + msg = _group_message(text, entities=[SimpleNamespace(type="mention", offset=3, length=11)]) + elif trigger == "code": + # Telegram says this is code, not a mention; a reply admits the turn. + msg = _group_message("@hermes_bot", reply_to_bot=True, entities=[SimpleNamespace(type="code", offset=0, length=11)]) + else: + msg = _group_message("wake hello" if trigger == "wake_word" else "hello", reply_to_bot=trigger == "reply") + if msg.reply_to_message: + for attr in ("photo", "video", "voice", "audio", "document"): + setattr(msg.reply_to_message, attr, None) + update = SimpleNamespace(update_id=1002, message=msg, effective_message=None) + handler = adapter._handle_command if msg_type == MessageType.COMMAND else adapter._handle_text_message + await handler(update, SimpleNamespace()) + + assert len(events) == 1 + event = events[0] + if trigger == "dm": + assert not event.channel_prompt + else: + expected = "yes" if trigger in {"mention", "text_mention", "command"} else "no" + assert f"Current message explicitly mentions you: {expected}" in event.channel_prompt + assert event.text == ("/new" if trigger == "command" else msg.text) + assert event.source.user_id == "111" + + asyncio.run(run()) diff --git a/website/docs/user-guide/messaging/telegram.md b/website/docs/user-guide/messaging/telegram.md index 2becfab4ca..6d1e2a19b7 100644 --- a/website/docs/user-guide/messaging/telegram.md +++ b/website/docs/user-guide/messaging/telegram.md @@ -578,6 +578,8 @@ telegram: With this setup, a group message like `@research_bot @ops_bot summarize this` is processed by `research_bot` and `ops_bot` only. Other Hermes bots in the group stay silent, even if the message is a reply to one of their earlier messages or would otherwise match a shared wake word. +Group conversation text and media captions retain all mentions, including the receiving bot's own handle. Each turn also includes the bot's Telegram username and whether the original message explicitly mentioned it. This preserves multi-bot addressing without treating replies, wake words, or open-group messages as explicit mentions. Slash commands still use the normal command-trigger cleanup. + Set `exclusive_bot_mentions: false` only for legacy groups where explicit mentions should not override reply and wake-word triggers. To operate several profiles, run the gateway command once per profile. For example: From bc31e04d83ee95556ffb6c667739c485f2f1e74f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 04:32:29 -0700 Subject: [PATCH 123/276] fix(telegram): strip our handle only as sole addressee; keep the identity line session-stable MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two corrections on top of the mention-preservation pick: - Stripping our own trigger handle from every group message was replaced by never stripping it in groups, which broke the pre-existing contract that `@bot 2` answers a pending clarify prompt and `@bot ok` approves (the intercepts match the exact stripped text). Strip only when no other participant is named; when other bots/users are mentioned the text is kept verbatim so `@research_bot , @ops_bot …` no longer arrives as `, @ops_bot …`. - The addressing block carried a per-message fact ("explicitly mentions you: yes/no") inside `channel_prompt`, which is part of the cached-agent signature: a mention turn followed by a reply turn rebuilt the AIAgent (prompt-cache miss) every time the shape flipped. Keep only the username line, byte-identical for the life of the session. Tests rewritten to pin both contracts (sole-addressee stripping; identical channel_prompt and agent signature across a mention turn and a reply turn). --- plugins/platforms/telegram/adapter.py | 16 +++--- .../platforms/telegram/telegram_context.py | 54 ++++++++++++++----- .../gateway/test_telegram_mention_context.py | 45 +++++++++++----- website/docs/user-guide/messaging/telegram.md | 2 +- 4 files changed, 81 insertions(+), 36 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 555ba218cf..bf29b640cc 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -5657,10 +5657,11 @@ class TelegramAdapter(BasePlatformAdapter): return False async def _build_triggered_event(self, msg, update, msg_type: MessageType) -> MessageEvent: - """Keep group conversation addressing; normalize command triggers for dispatch.""" + """Event for an addressed text/command: trigger text cleaned (sole addressee only), replied-to + media cached, attribution applied.""" + from plugins.platforms.telegram.telegram_context import group_trigger_text event = self._build_message_event(msg, msg_type, update_id=update.update_id) - if msg_type == MessageType.COMMAND or not self._is_group_chat(msg): - event.text = self._clean_bot_trigger_text(event.text) + event.text = group_trigger_text(self, msg, event.text) await self._cache_replied_media(msg, event) return self._apply_telegram_group_observe_attribution(event) @@ -5979,13 +5980,14 @@ class TelegramAdapter(BasePlatformAdapter): if self._should_observe_unmentioned_group_message(msg): _event = self._build_message_event(msg, self._media_message_type(msg), update_id=update.update_id) if msg.caption: - _event.text = msg.caption + _event.text = self._clean_bot_trigger_text(msg.caption) await self._cache_observed_media(msg, _event) self._observe_unmentioned_group_message(msg, _event.message_type, update_id=update.update_id, event=_event) return event = self._build_message_event(msg, self._media_message_type(msg), update_id=update.update_id) if msg.caption: - event.text = msg.caption if self._is_group_chat(msg) else self._clean_bot_trigger_text(msg.caption) + from plugins.platforms.telegram.telegram_context import group_trigger_text + event.text = group_trigger_text(self, msg, msg.caption) # Stickers: _handle_sticker overwrites event.text with its vision description, so observe attribution must run after it. if msg.sticker: await self._handle_sticker(msg, event) @@ -6282,14 +6284,14 @@ class TelegramAdapter(BasePlatformAdapter): is_bot=bool(getattr(user, "is_bot", False)) if user else False) reply_to_id, reply_to_text = self._reply_context(message) from gateway.platforms.base import resolve_channel_prompt # per-channel/topic ephemeral prompt - from plugins.platforms.telegram.telegram_context import group_addressing_prompt + from plugins.platforms.telegram.telegram_context import group_identity_prompt _chat_id_str = str(chat.id) channel_prompt = resolve_channel_prompt(self.config.extra, thread_id_str or _chat_id_str, _chat_id_str if thread_id_str else None) return MessageEvent( text=message.text or "", message_type=msg_type, source=source, raw_message=message, message_id=str(message.message_id), platform_update_id=update_id, reply_to_message_id=reply_to_id, reply_to_text=reply_to_text, auto_skill=topic_skill, - channel_prompt=group_addressing_prompt(self, message, channel_prompt), + channel_prompt=group_identity_prompt(self, message, channel_prompt), timestamp=message.date) # -- Message reactions (processing lifecycle) -- diff --git a/plugins/platforms/telegram/telegram_context.py b/plugins/platforms/telegram/telegram_context.py index a4ab9ada1b..b02f9ca313 100644 --- a/plugins/platforms/telegram/telegram_context.py +++ b/plugins/platforms/telegram/telegram_context.py @@ -1,27 +1,53 @@ -"""Per-message Telegram addressing facts, outside the cached session prompt.""" +"""Telegram group addressing helpers, kept out of the adapter facade. -from typing import TYPE_CHECKING +The identity line lives in ``channel_prompt`` and therefore in the cached-agent signature: it +must be stable for the life of a session (username only — never a per-message fact). +""" + +from typing import TYPE_CHECKING, Optional if TYPE_CHECKING: from telegram import Message from plugins.platforms.telegram.adapter import TelegramAdapter -def group_addressing_prompt( - adapter: "TelegramAdapter", message: "Message", channel_prompt: str | None, -) -> str | None: +def mentions_other_participants(adapter: "TelegramAdapter", message: "Message") -> bool: + """True when a ``mention``/``text_mention`` entity names someone other than this bot.""" + own = adapter._current_bot_username() + bot_id = getattr(adapter._bot, "id", None) if adapter._bot else None + for source_text, entities in adapter._entity_sources(message): + for entity in entities: + entity_type = adapter._entity_type(entity) + if entity_type == "mention": + handle = (adapter._entity_span(source_text, entity) or "").strip().lstrip("@").lower() + if handle and handle != own: + return True + elif entity_type == "text_mention": + user = getattr(entity, "user", None) + if user is not None and getattr(user, "id", None) != bot_id: + return True + return False + + +def group_trigger_text(adapter: "TelegramAdapter", message: "Message", text: Optional[str]) -> Optional[str]: + """Strip our own handle only when we are the sole addressee. With other participants named, + ``@research_bot , @ops_bot are you both listening?`` must not reach us as ``, @ops_bot …``.""" + if adapter._is_group_chat(message) and mentions_other_participants(adapter, message): + return text + return adapter._clean_bot_trigger_text(text) + + +def group_identity_prompt( + adapter: "TelegramAdapter", message: "Message", channel_prompt: Optional[str], +) -> Optional[str]: + """Session-stable identity line so the model can read retained @mentions as itself or not.""" if not adapter._is_group_chat(message) or not getattr(adapter, "_bot", None): return channel_prompt username = adapter._current_bot_username() if not username: return channel_prompt - # Use the same entity-aware check as admission, not the cleaned text or the - # fact that a turn was dispatched (replies/wake words/open groups also pass). - mentioned = "yes" if adapter._message_mentions_bot(message) else "no" - addressing = ( - "Telegram addressing context (current message only):\n" - f"- Your Telegram bot username: @{username}\n" - f"- Current message explicitly mentions you: {mentioned}\n" - "Mentions of other bots do not by themselves ask you to relay the message." + identity = ( + f"Your Telegram bot username in this group: @{username}. " + "Mentions of other bots are not requests for you to relay the message." ) - return f"{channel_prompt}\n\n{addressing}" if channel_prompt else addressing + return f"{channel_prompt}\n\n{identity}" if channel_prompt else identity diff --git a/tests/gateway/test_telegram_mention_context.py b/tests/gateway/test_telegram_mention_context.py index 35c36985f0..b20b6b63e5 100644 --- a/tests/gateway/test_telegram_mention_context.py +++ b/tests/gateway/test_telegram_mention_context.py @@ -1,4 +1,5 @@ -"""Addressing information must survive routing into the agent's event.""" +"""Multi-bot addressing must survive routing into the agent's event without destabilising the +session prompt.""" import asyncio from types import SimpleNamespace @@ -7,12 +8,19 @@ from unittest.mock import AsyncMock import pytest from gateway.platforms.base import MessageType +from gateway.run import GatewayRunner from tests.gateway.test_telegram_group_gating import ( _dm_message, _group_message, _group_voice_message, _make_adapter, _mention_entities, ) +def _prompt_signature(event): + return GatewayRunner._agent_config_signature( + "model", {"api_key": "k", "base_url": "u", "provider": "p"}, ["messaging"], event.channel_prompt or "", + ) + + @pytest.mark.parametrize("media", [False, True]) @pytest.mark.parametrize("observe", [False, True]) def test_multi_bot_addressing_survives_real_handlers(media, observe): @@ -48,8 +56,7 @@ def test_multi_bot_addressing_survives_real_handlers(media, observe): assert len(events) == 1 event = events[0] assert text in event.text # Preserve both recipients and their positions. - assert f"Your Telegram bot username: @{username}" in event.channel_prompt - assert "Current message explicitly mentions you: yes" in event.channel_prompt + assert f"@{username}" in event.channel_prompt assert "Keep answers concise." in event.channel_prompt assert ("observed Telegram group context" in event.channel_prompt) == observe assert event.source.user_id == (None if observe else "111") @@ -58,7 +65,10 @@ def test_multi_bot_addressing_survives_real_handlers(media, observe): @pytest.mark.parametrize("trigger", ["mention", "text_mention", "reply", "wake_word", "open", "code", "command", "dm"]) -def test_addressing_context_reports_original_entities_without_inventing_mentions(trigger): +def test_sole_addressee_text_stays_clean_and_prompt_is_session_stable(trigger): + """Our own handle is still stripped when nobody else is named (clarify answers like ``@bot 2`` + keep resolving), and the identity block is identical across turns: it rides the cached-agent + signature, so a per-message fact there would rebuild the agent every turn.""" async def run(): adapter = _make_adapter(require_mention=trigger != "open", mention_patterns=["^wake\\b"]) adapter._ensure_forum_commands = AsyncMock() @@ -74,7 +84,7 @@ def test_addressing_context_reports_original_entities_without_inventing_mentions elif trigger == "text_mention": msg = _group_message("Hermes hello", entities=[SimpleNamespace(type="text_mention", offset=0, length=6, user=SimpleNamespace(id=999))]) elif trigger == "mention": - text = "😀 @hermes_bot hello" + text = "😀 @hermes_bot 2" msg = _group_message(text, entities=[SimpleNamespace(type="mention", offset=3, length=11)]) elif trigger == "code": # Telegram says this is code, not a mention; a reply admits the turn. @@ -84,18 +94,25 @@ def test_addressing_context_reports_original_entities_without_inventing_mentions if msg.reply_to_message: for attr in ("photo", "video", "voice", "audio", "document"): setattr(msg.reply_to_message, attr, None) - update = SimpleNamespace(update_id=1002, message=msg, effective_message=None) handler = adapter._handle_command if msg_type == MessageType.COMMAND else adapter._handle_text_message - await handler(update, SimpleNamespace()) + await handler(SimpleNamespace(update_id=1002, message=msg, effective_message=None), SimpleNamespace()) + # Second turn in the same chat with a different addressing shape (reply, no entities). + follow_up = _dm_message("thanks") if trigger == "dm" else _group_message("thanks", reply_to_bot=True) + if follow_up.reply_to_message: + for attr in ("photo", "video", "voice", "audio", "document"): + setattr(follow_up.reply_to_message, attr, None) + await adapter._handle_text_message(SimpleNamespace(update_id=1003, message=follow_up, effective_message=None), SimpleNamespace()) - assert len(events) == 1 - event = events[0] + assert len(events) == 2 + first, second = events + expected_text = {"command": "/new", "mention": "😀 2", "code": "@hermes_bot"}.get(trigger, msg.text) + assert first.text == expected_text if trigger == "dm": - assert not event.channel_prompt + assert not first.channel_prompt else: - expected = "yes" if trigger in {"mention", "text_mention", "command"} else "no" - assert f"Current message explicitly mentions you: {expected}" in event.channel_prompt - assert event.text == ("/new" if trigger == "command" else msg.text) - assert event.source.user_id == "111" + assert "@hermes_bot" in first.channel_prompt + assert first.channel_prompt == second.channel_prompt + assert _prompt_signature(first) == _prompt_signature(second) + assert first.source.user_id == "111" asyncio.run(run()) diff --git a/website/docs/user-guide/messaging/telegram.md b/website/docs/user-guide/messaging/telegram.md index 6d1e2a19b7..deb60b4e72 100644 --- a/website/docs/user-guide/messaging/telegram.md +++ b/website/docs/user-guide/messaging/telegram.md @@ -578,7 +578,7 @@ telegram: With this setup, a group message like `@research_bot @ops_bot summarize this` is processed by `research_bot` and `ops_bot` only. Other Hermes bots in the group stay silent, even if the message is a reply to one of their earlier messages or would otherwise match a shared wake word. -Group conversation text and media captions retain all mentions, including the receiving bot's own handle. Each turn also includes the bot's Telegram username and whether the original message explicitly mentioned it. This preserves multi-bot addressing without treating replies, wake words, or open-group messages as explicit mentions. Slash commands still use the normal command-trigger cleanup. +Group conversation text and media captions keep every mention when the message names other participants too (`@research_bot , @ops_bot are you both listening?` reaches `research_bot` verbatim); when this bot is the only one addressed, its own handle is still stripped so short answers such as `@hermes_bot 2` keep working. Group turns also carry the bot's own Telegram username in the per-channel context so the model can tell which retained mentions are for it. Slash commands still use the normal command-trigger cleanup. Set `exclusive_bot_mentions: false` only for legacy groups where explicit mentions should not override reply and wake-word triggers. From 33cefdd8e51ae9a2f66cb2b6f4ef6dd033abc113 Mon Sep 17 00:00:00 2001 From: Xipong <217837358+Xipong@users.noreply.github.com> Date: Sun, 2 Aug 2026 05:20:19 +0300 Subject: [PATCH 124/276] fix(fallback): restore primary route after picker errors --- hermes_cli/fallback_cmd.py | 85 +++++++++------ tests/hermes_cli/test_fallback_cmd.py | 146 +++++++++++++++++++++++++- 2 files changed, 196 insertions(+), 35 deletions(-) diff --git a/hermes_cli/fallback_cmd.py b/hermes_cli/fallback_cmd.py index 79a3d3d927..6867919dff 100644 --- a/hermes_cli/fallback_cmd.py +++ b/hermes_cli/fallback_cmd.py @@ -13,6 +13,9 @@ from hermes_cli.fallback_config import get_fallback_chain _read_chain = get_fallback_chain +_MISSING_ACTIVE_PROVIDER = object() + + def _identity(entry: Dict[str, Any]): """BackendIdentity for a ``{provider, model, base_url?}`` entry.""" from agent.backend_identity import BackendIdentity @@ -45,23 +48,25 @@ def _extract_fallback_from_model_cfg(model_cfg: Any) -> Optional[Dict[str, Any]] def _snapshot_auth_active_provider() -> Any: - """Current ``active_provider`` in auth.json, or None if unavailable.""" - try: - from hermes_cli.auth import _load_auth_store - return _load_auth_store().get("active_provider") - except Exception: - return None + """Return the current ``active_provider`` in auth.json.""" + from hermes_cli.auth import _auth_store_lock, _load_auth_store + + with _auth_store_lock(): + store = _load_auth_store() + return store.get("active_provider", _MISSING_ACTIVE_PROVIDER) def _restore_auth_active_provider(value: Any) -> None: - """Write back a snapshotted ``active_provider``; best-effort (user re-runs `hermes model`), never fails the add.""" - try: - from hermes_cli.auth import _auth_store_lock, _load_auth_store, _save_auth_store - with _auth_store_lock(): - _save_auth_store({**_load_auth_store(), "active_provider": value}) - except Exception: - pass + """Write back a previously snapshotted ``active_provider`` value.""" + from hermes_cli.auth import _auth_store_lock, _load_auth_store, _save_auth_store + with _auth_store_lock(): + store = _load_auth_store() + if value is _MISSING_ACTIVE_PROVIDER: + store.pop("active_provider", None) + else: + store["active_provider"] = value + _save_auth_store(store) def _restore_model_cfg(model_before: Any) -> None: """Restore ``config["model"]`` to a previously-captured snapshot.""" @@ -73,6 +78,21 @@ def _restore_model_cfg(model_before: Any) -> None: save_config(cfg) +def _restore_primary_route(model_before: Any, active_provider_before: Any) -> None: + """Attempt both halves of temporary picker-route restoration.""" + errors: list[BaseException] = [] + try: + _restore_model_cfg(model_before) + except BaseException as exc: + errors.append(exc) + try: + _restore_auth_active_provider(active_provider_before) + except BaseException as exc: + errors.append(exc) + if errors: + details = "; ".join(str(exc) for exc in errors) + raise RuntimeError(f"Could not fully restore the primary route: {details}") from errors[0] + def _entries(n: int) -> str: return f"{n} {'entry' if n == 1 else 'entries'}" @@ -122,45 +142,43 @@ def cmd_fallback_add(args) -> None: from hermes_cli.config import load_config, save_config _require_tty("fallback add") - # Snapshot BEFORE the picker runs: "picked" vs "cancelled" is decided by comparing before/after, - # and the primary must be restored either way. + # Snapshot BEFORE the picker runs; both route stores must be restored on every exit path. model_before = copy.deepcopy(load_config().get("model")) active_provider_before = _snapshot_auth_active_provider() print("\n Adding a fallback provider. The picker below is the same one used by\n" " `hermes model` — select the provider + model you want as a fallback.\n") - def _restore() -> None: - _restore_model_cfg(model_before) - - _restore_auth_active_provider(active_provider_before) try: select_provider_and_model(args=args) - except SystemExit: # some provider flows exit on auth failure — restore state and re-raise - _restore() + after_cfg = load_config() + model_after = after_cfg.get("model") + new_entry = _extract_fallback_from_model_cfg(model_after) + except BaseException as picker_error: + try: + _restore_primary_route(model_before, active_provider_before) + except Exception as restore_error: + picker_error.add_note( + "Could not fully restore the primary route after fallback " + f"selection failed: {restore_error}" + ) raise - new_entry = _extract_fallback_from_model_cfg(load_config().get("model")) - if not new_entry: # picker didn't complete (user cancelled or flow bailed) - _restore() + + # From here onward no identity/import/append failure can strand the temporary picker route. + _restore_primary_route(model_before, active_provider_before) + + if not new_entry: print("\n No fallback added.") return - # Same deployment as the primary → nothing to add. Identity semantics are owned by - # agent.backend_identity: same provider+model on a DIFFERENT explicit base_url is a different - # backend (multi-endpoint pool) and a legitimate fallback. - # Picker picked the same thing that's already the primary → nothing changed, and there's nothing useful - # to add as a fallback to itself. See #54250, #57584, #62984. from agent.backend_identity import same_deployment new_ident = _identity(new_entry) primary_entry = _extract_fallback_from_model_cfg(model_before) if primary_entry and same_deployment(_identity(primary_entry), new_ident): - _restore() print(f"\n Selected model matches the current primary ({_format_entry(new_entry)}).") print(" A provider cannot be a fallback for itself — no change.") return - # Restore the primary, then re-load (rather than mutating the post-picker config) because the - # picker may have touched other top-level keys (custom_providers, credentials) we want to keep. - _restore() + # Reload after primary restoration; picker-created providers/credentials remain. final_cfg = load_config() chain = _read_chain(final_cfg) if any(same_deployment(_identity(existing), new_ident) for existing in chain): @@ -173,7 +191,6 @@ def cmd_fallback_add(args) -> None: print(f" Chain is now {_entries(len(chain))} long.\n") print(" Run `hermes fallback list` to view, or `hermes fallback remove` to delete.") - def cmd_fallback_remove(args) -> None: # noqa: ARG001 """Pick an entry from the chain and remove it.""" from hermes_cli.config import save_config diff --git a/tests/hermes_cli/test_fallback_cmd.py b/tests/hermes_cli/test_fallback_cmd.py index 8f1169d66c..3e0393f282 100644 --- a/tests/hermes_cli/test_fallback_cmd.py +++ b/tests/hermes_cli/test_fallback_cmd.py @@ -15,7 +15,6 @@ import yaml @pytest.fixture() def isolated_home(tmp_path, monkeypatch): - monkeypatch.setattr(Path, "home", lambda: tmp_path) home = tmp_path / ".hermes" home.mkdir(exist_ok=True) monkeypatch.setenv("HERMES_HOME", str(home)) @@ -118,6 +117,20 @@ class TestListCommand: # --------------------------------------------------------------------------- class TestAddCommand: + def test_auth_snapshot_failure_aborts_before_picker(self): + from hermes_cli import fallback_cmd + + picker = object() + with patch( + "hermes_cli.auth._load_auth_store", + side_effect=OSError("auth read failed"), + ), patch( + "hermes_cli.main.select_provider_and_model", + picker, + ), patch("hermes_cli.main._require_tty"): + with pytest.raises(OSError, match="auth read failed"): + fallback_cmd.cmd_fallback_add(types.SimpleNamespace()) + def test_add_appends_new_entry(self, isolated_home, capsys): _write_config(isolated_home, { "model": {"provider": "anthropic", "default": "claude-sonnet-4-6"}, @@ -199,6 +212,13 @@ class TestAddCommand: "base_url": "https://openrouter.ai/api/v1", "api_mode": "chat_completions", } + cfg["custom_providers"] = [ + { + "name": "Picker-created endpoint", + "base_url": "https://picker.example/v1", + "model": "picker-model", + } + ] save_config(cfg) with patch("hermes_cli.main.select_provider_and_model", side_effect=fake_picker), \ @@ -215,6 +235,130 @@ class TestAddCommand: # Fallback added assert len(cfg["fallback_providers"]) == 1 assert cfg["fallback_providers"][0]["provider"] == "openrouter" + assert cfg["custom_providers"] == [ + { + "name": "Picker-created endpoint", + "base_url": "https://picker.example/v1", + "model": "picker-model", + } + ] + + def test_post_picker_config_read_failure_restores_route(self): + from hermes_cli import fallback_cmd + + primary_model = { + "provider": "anthropic", + "default": "claude-sonnet-4-6", + } + post_read_error = OSError("post-picker config read failed") + reads = iter([{"model": primary_model}, post_read_error]) + + def load_config(): + value = next(reads) + if isinstance(value, BaseException): + raise value + return value + + restore_calls = [] + with patch("hermes_cli.config.load_config", side_effect=load_config), patch( + "hermes_cli.main._require_tty" + ), patch("hermes_cli.main.select_provider_and_model"), patch.object( + fallback_cmd, "_snapshot_auth_active_provider", return_value="old-provider" + ), patch.object( + fallback_cmd, + "_restore_primary_route", + side_effect=lambda model, provider: restore_calls.append((model, provider)), + ): + with pytest.raises(OSError) as exc_info: + fallback_cmd.cmd_fallback_add(types.SimpleNamespace()) + + assert exc_info.value is post_read_error + assert restore_calls == [(primary_model, "old-provider")] + + @pytest.mark.parametrize( + ("failure_type", "message"), + [ + (OSError, "config write failed"), + (KeyboardInterrupt, "config restore interrupted"), + ], + ) + def test_restore_attempts_auth_after_model_restore_failure( + self, failure_type, message + ): + from hermes_cli import fallback_cmd + + auth_calls = [] + with patch.object( + fallback_cmd, + "_restore_model_cfg", + side_effect=failure_type(message), + ), patch.object( + fallback_cmd, + "_restore_auth_active_provider", + side_effect=lambda value: auth_calls.append(value), + ): + with pytest.raises(RuntimeError, match=message): + fallback_cmd._restore_primary_route("old-model", "old-provider") + + assert auth_calls == ["old-provider"] + + def test_restore_preserves_absent_active_provider(self): + from contextlib import nullcontext + + from hermes_cli import auth, fallback_cmd + + store = {"version": 1, "providers": {}} + + def save_auth(value): + store.clear() + store.update(value) + + with patch.object(auth, "_load_auth_store", lambda: dict(store)), patch.object( + auth, "_save_auth_store", save_auth + ), patch.object(auth, "_auth_store_lock", nullcontext): + before = fallback_cmd._snapshot_auth_active_provider() + fallback_cmd._restore_auth_active_provider(before) + + assert "active_provider" not in store + + def test_picker_failure_restores_persisted_primary_without_masking_error( + self, isolated_home + ): + from hermes_cli import fallback_cmd + + primary_model = { + "provider": "anthropic", + "default": "claude-sonnet-4-6", + "base_url": "https://api.anthropic.com", + "api_mode": "anthropic_messages", + } + _write_config(isolated_home, {"model": primary_model, "theme": "midnight"}) + picker_error = LookupError("picker failed") + + def failing_picker(args=None): + from hermes_cli.config import load_config, save_config + + cfg = load_config() + cfg["model"] = { + "provider": "openrouter", + "default": "anthropic/claude-sonnet-4.6", + "base_url": "https://openrouter.ai/api/v1", + "api_mode": "chat_completions", + } + save_config(cfg) + raise picker_error + + with patch( + "hermes_cli.main.select_provider_and_model", + side_effect=failing_picker, + ), patch("hermes_cli.main._require_tty"): + with pytest.raises(LookupError, match="picker failed") as exc_info: + fallback_cmd.cmd_fallback_add(types.SimpleNamespace()) + + assert exc_info.value is picker_error + persisted = _read_config(isolated_home) + assert persisted["model"] == primary_model + assert persisted["theme"] == "midnight" # --------------------------------------------------------------------------- From 91b4d9fb67f5cb5bd04dacd20c32f73d68ebff3b Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:05:36 -0700 Subject: [PATCH 125/276] test(fallback): keep the two persisted-config invariants; PTY harness for the picker error path Trim the salvaged suite to the behaviour contracts: a picker exception or Ctrl+C leaves config.yaml's model exactly as snapshotted (parametrized), and an absent active_provider stays absent through snapshot+restore. The mock-dispatch tests (asserting calls into our own helpers) are dropped. evals/cli_fallback_add_picker_error.py drives the real `hermes fallback add` under a Linux PTY into an ordinary picker OSError and checks the persisted model: stranded on base, restored after. --- evals/cli_fallback_add_picker_error.py | 135 +++++++++++++++++++++++++ hermes_cli/fallback_cmd.py | 3 + tests/hermes_cli/test_fallback_cmd.py | 82 ++------------- 3 files changed, 144 insertions(+), 76 deletions(-) create mode 100644 evals/cli_fallback_add_picker_error.py diff --git a/evals/cli_fallback_add_picker_error.py b/evals/cli_fallback_add_picker_error.py new file mode 100644 index 0000000000..da90b18782 --- /dev/null +++ b/evals/cli_fallback_add_picker_error.py @@ -0,0 +1,135 @@ +"""Drive the real ``hermes fallback add`` under a Linux PTY into an ordinary picker error. + +Run from the checkout with the project Python. No provider requests are sent: the picker +target is a saved custom provider with ``discover_models: false``. The auth store is made +unreadable AFTER the provider menu renders (the pre-picker snapshot has already happened), so +the canonical picker writes the temporary primary route to config.yaml and then fails inside +``deactivate_provider`` with a plain ``PermissionError`` -- not ``SystemExit``. + +The invariant under test: config.yaml ``model`` must equal the pre-picker primary afterwards. +""" +import argparse +import errno +import json +import os +from pathlib import Path +import pty +import re +import select +import signal +import struct +import subprocess +import sys +import tempfile +import termios +import time +import fcntl + +PRIMARY = {"provider": "openrouter", "default": "primary/model-a", + "base_url": "https://openrouter.ai/api/v1", "api_mode": "chat_completions"} +CONFIG = ( + "model:\n provider: openrouter\n default: primary/model-a\n" + " base_url: https://openrouter.ai/api/v1\n api_mode: chat_completions\n" + "custom_providers:\n - name: LocalLab\n base_url: http://127.0.0.1:9/v1\n" + " model: lab-model\n discover_models: false\n models:\n - lab-model\n" + "memory:\n provider: ''\n") + + +def _persisted_model(root: Path, env: dict) -> dict: + """``config.yaml`` ``model`` section as the CLI itself reads it (owner module, same env).""" + out = subprocess.run( + [sys.executable, "-c", "import json; from hermes_cli.config import load_config; " + "print(json.dumps(load_config().get('model')))"], + cwd=root, env=env, capture_output=True, text=True, check=True) + return json.loads(out.stdout.strip().splitlines()[-1]) + + +def run(root: Path, output: Path) -> dict: + with tempfile.TemporaryDirectory(prefix="hermes_test_fallback_") as home: + hh = Path(home) / ".hermes" + hh.mkdir() + (hh / "config.yaml").write_text(CONFIG, encoding="utf-8") + (hh / ".env").write_text("OPENROUTER_API_KEY=local-not-used\n", encoding="utf-8") + auth = hh / "auth.json" + auth.write_text(json.dumps({"version": 1, "providers": {}, "active_provider": "nous"})) + # A stub ``curses`` package forces every menu onto its numbered fallback so the PTY + # exchange is line-oriented (the curses UI is not what is under test here). + shim = Path(home) / "shim" / "curses" + shim.mkdir(parents=True) + (shim / "__init__.py").write_text("raise ImportError('curses disabled for PTY harness')\n") + env = {"PATH": os.environ["PATH"], "HOME": home, "HERMES_HOME": str(hh), + "PYTHONPATH": f"{shim.parent}{os.pathsep}{root}", "PYTHONUNBUFFERED": "1", + "TERM": "dumb", "LANG": "C.UTF-8"} + master, slave = pty.openpty() + fcntl.ioctl(slave, termios.TIOCSWINSZ, struct.pack("HHHH", 50, 120, 0, 0)) + proc = subprocess.Popen([sys.executable, "-m", "hermes_cli.main", "fallback", "add"], + cwd=root, env=env, stdin=slave, stdout=slave, stderr=slave, + start_new_session=True) + os.close(slave) + data = bytearray() + + def pump_until(predicate, timeout=60): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if predicate(bytes(data)): + return True + if select.select([master], [], [], 0.1)[0]: + try: + chunk = os.read(master, 65536) + except OSError as exc: + if exc.errno == errno.EIO: + return predicate(bytes(data)) + raise + if not chunk: + return predicate(bytes(data)) + data.extend(chunk) + return predicate(bytes(data)) + + try: + assert pump_until(lambda b: b"Choice [default" in b), data[-2000:] + text = bytes(data).decode(errors="replace") + row = re.search(r"(\d+)\. LocalLab", text) + assert row, text[-3000:] + # Snapshot is done (menu is up); now make the auth store unreadable so the picker's + # own deactivate_provider() fails with an ordinary OSError after writing the model. + auth.chmod(0) + os.write(master, f"{row.group(1)}\r".encode()) + offset = len(data) + assert pump_until(lambda b: b"Choice [" in b[offset:]), data[-2000:] + os.write(master, b"1\r") + exited = pump_until(lambda b: proc.poll() is not None, 90) + proc.wait(timeout=30) + auth.chmod(0o600) + model_after = _persisted_model(root, env) + text = bytes(data).decode(errors="replace") + return {"exited": exited, "returncode": proc.returncode, + "picker_error_surfaced": "PermissionError" in text, + "model_after": model_after, "primary_restored": model_after == PRIMARY, + "auth_active_provider": json.loads(auth.read_text()).get("active_provider"), + "restore_note": "Could not fully restore" in text, + "raw_path": str(output / "fallback-add-picker-error.pty")} + finally: + output.mkdir(parents=True, exist_ok=True) + (output / "fallback-add-picker-error.pty").write_bytes(data) + if proc.poll() is None: + os.killpg(proc.pid, signal.SIGKILL) + proc.wait(timeout=30) + os.close(master) + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--root", required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--expect", choices=("stranded", "restored"), required=True) + args = parser.parse_args() + result = run(Path(args.root).resolve(), args.output) + (args.output / "results.json").write_text(json.dumps(result, indent=2) + "\n") + print(json.dumps(result, indent=2)) + assert result["exited"] and result["returncode"] != 0, result + assert result["picker_error_surfaced"], result + assert result["primary_restored"] == (args.expect == "restored"), result + + +if __name__ == "__main__": + main() diff --git a/hermes_cli/fallback_cmd.py b/hermes_cli/fallback_cmd.py index 6867919dff..866fb88d9a 100644 --- a/hermes_cli/fallback_cmd.py +++ b/hermes_cli/fallback_cmd.py @@ -68,6 +68,7 @@ def _restore_auth_active_provider(value: Any) -> None: store["active_provider"] = value _save_auth_store(store) + def _restore_model_cfg(model_before: Any) -> None: """Restore ``config["model"]`` to a previously-captured snapshot.""" from hermes_cli.config import load_config, save_config @@ -93,6 +94,7 @@ def _restore_primary_route(model_before: Any, active_provider_before: Any) -> No details = "; ".join(str(exc) for exc in errors) raise RuntimeError(f"Could not fully restore the primary route: {details}") from errors[0] + def _entries(n: int) -> str: return f"{n} {'entry' if n == 1 else 'entries'}" @@ -191,6 +193,7 @@ def cmd_fallback_add(args) -> None: print(f" Chain is now {_entries(len(chain))} long.\n") print(" Run `hermes fallback list` to view, or `hermes fallback remove` to delete.") + def cmd_fallback_remove(args) -> None: # noqa: ARG001 """Pick an entry from the chain and remove it.""" from hermes_cli.config import save_config diff --git a/tests/hermes_cli/test_fallback_cmd.py b/tests/hermes_cli/test_fallback_cmd.py index 3e0393f282..a20a85293b 100644 --- a/tests/hermes_cli/test_fallback_cmd.py +++ b/tests/hermes_cli/test_fallback_cmd.py @@ -117,20 +117,6 @@ class TestListCommand: # --------------------------------------------------------------------------- class TestAddCommand: - def test_auth_snapshot_failure_aborts_before_picker(self): - from hermes_cli import fallback_cmd - - picker = object() - with patch( - "hermes_cli.auth._load_auth_store", - side_effect=OSError("auth read failed"), - ), patch( - "hermes_cli.main.select_provider_and_model", - picker, - ), patch("hermes_cli.main._require_tty"): - with pytest.raises(OSError, match="auth read failed"): - fallback_cmd.cmd_fallback_add(types.SimpleNamespace()) - def test_add_appends_new_entry(self, isolated_home, capsys): _write_config(isolated_home, { "model": {"provider": "anthropic", "default": "claude-sonnet-4-6"}, @@ -243,65 +229,6 @@ class TestAddCommand: } ] - def test_post_picker_config_read_failure_restores_route(self): - from hermes_cli import fallback_cmd - - primary_model = { - "provider": "anthropic", - "default": "claude-sonnet-4-6", - } - post_read_error = OSError("post-picker config read failed") - reads = iter([{"model": primary_model}, post_read_error]) - - def load_config(): - value = next(reads) - if isinstance(value, BaseException): - raise value - return value - - restore_calls = [] - with patch("hermes_cli.config.load_config", side_effect=load_config), patch( - "hermes_cli.main._require_tty" - ), patch("hermes_cli.main.select_provider_and_model"), patch.object( - fallback_cmd, "_snapshot_auth_active_provider", return_value="old-provider" - ), patch.object( - fallback_cmd, - "_restore_primary_route", - side_effect=lambda model, provider: restore_calls.append((model, provider)), - ): - with pytest.raises(OSError) as exc_info: - fallback_cmd.cmd_fallback_add(types.SimpleNamespace()) - - assert exc_info.value is post_read_error - assert restore_calls == [(primary_model, "old-provider")] - - @pytest.mark.parametrize( - ("failure_type", "message"), - [ - (OSError, "config write failed"), - (KeyboardInterrupt, "config restore interrupted"), - ], - ) - def test_restore_attempts_auth_after_model_restore_failure( - self, failure_type, message - ): - from hermes_cli import fallback_cmd - - auth_calls = [] - with patch.object( - fallback_cmd, - "_restore_model_cfg", - side_effect=failure_type(message), - ), patch.object( - fallback_cmd, - "_restore_auth_active_provider", - side_effect=lambda value: auth_calls.append(value), - ): - with pytest.raises(RuntimeError, match=message): - fallback_cmd._restore_primary_route("old-model", "old-provider") - - assert auth_calls == ["old-provider"] - def test_restore_preserves_absent_active_provider(self): from contextlib import nullcontext @@ -321,9 +248,13 @@ class TestAddCommand: assert "active_provider" not in store + @pytest.mark.parametrize("picker_error", [LookupError("picker failed"), KeyboardInterrupt()], + ids=["exception", "ctrl-c"]) def test_picker_failure_restores_persisted_primary_without_masking_error( - self, isolated_home + self, isolated_home, picker_error ): + """An ordinary picker exception or a Ctrl+C mid-picker must leave config.yaml's + ``model`` exactly as it was before ``fallback add`` started (base only handled SystemExit).""" from hermes_cli import fallback_cmd primary_model = { @@ -333,7 +264,6 @@ class TestAddCommand: "api_mode": "anthropic_messages", } _write_config(isolated_home, {"model": primary_model, "theme": "midnight"}) - picker_error = LookupError("picker failed") def failing_picker(args=None): from hermes_cli.config import load_config, save_config @@ -352,7 +282,7 @@ class TestAddCommand: "hermes_cli.main.select_provider_and_model", side_effect=failing_picker, ), patch("hermes_cli.main._require_tty"): - with pytest.raises(LookupError, match="picker failed") as exc_info: + with pytest.raises(type(picker_error)) as exc_info: fallback_cmd.cmd_fallback_add(types.SimpleNamespace()) assert exc_info.value is picker_error From b0adce1cbfb9f3f7abb01ddef022cc231f3f3d8d Mon Sep 17 00:00:00 2001 From: tylman <1161649+tylman@users.noreply.github.com> Date: Wed, 2 Sep 2026 22:38:59 -0700 Subject: [PATCH 126/276] feat(models): add gemini-3.7-flash and gemini-3.8-flash support - Generalize Gemini 3 thinking config model prefix match to gemini-3* - Add pricing snapshot entries for gemini-3.7-flash and gemini-3.8-flash - Add model identifiers to Google, OpenRouter, and Vertex CLI lists - Add test coverage for gemini-3.8-flash thinkingLevel configuration --- agent/transports/chat_completions.py | 6 ++++-- agent/usage_pricing.py | 3 +++ hermes_cli/models_catalog_static.py | 6 ++++-- plugins/model-providers/openrouter/__init__.py | 2 +- tests/agent/transports/test_chat_completions.py | 14 ++++++++++++++ 5 files changed, 26 insertions(+), 5 deletions(-) diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 723174f87d..8171aec797 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -146,8 +146,10 @@ def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) -> return thinking_config if effort not in {"minimal", "low", "medium", "high", "xhigh", "max", "ultra"}: effort = "medium" - # Gemini 3 Flash documents low/medium/high; Gemini 3 Pro only low/high. - if normalized_model.startswith(("gemini-3", "gemini-3.1")): + # Gemini 3 Flash documents low/medium/high thinking levels; Gemini 3 Pro + # is stricter (low/high). Clamp Hermes' wider effort set to what each + # family accepts so we never forward an undocumented level verbatim. + if normalized_model.startswith("gemini-3"): if "flash" in normalized_model: thinking_config["thinkingLevel"] = ( "low" if effort in {"minimal", "low"} else "high" if effort in _HIGH_EFFORTS else "medium" diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index 0f73ce36f9..9830e2e28b 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -191,6 +191,9 @@ _SNAPSHOTS: tuple[tuple[str, Optional[str], str, dict], ...] = ( ("deepseek-chat", "deepseek-reasoner", "deepseek-v4-flash"): ("0.14", "0.28", "0.0028"), "deepseek-v4-pro": ("0.435", "0.87", "0.003625"), }), + ("google", "https://ai.google.dev/gemini-api/docs/pricing", "google-pricing-2026-09-02", { + ("gemini-3.8-flash", "gemini-3.7-flash"): ("0.75", "3.75", "0.075"), + }), ("google", "https://ai.google.dev/gemini-api/docs/pricing", "google-pricing-2026-07-28", { "gemini-3.6-flash": ("1.50", "7.50", "0.15"), "gemini-3.5-flash-lite": ("0.30", "2.50", "0.03"), }), diff --git a/hermes_cli/models_catalog_static.py b/hermes_cli/models_catalog_static.py index 97d96b9924..4772edb1ed 100644 --- a/hermes_cli/models_catalog_static.py +++ b/hermes_cli/models_catalog_static.py @@ -170,6 +170,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3-flash-preview", "gemini-2.5-pro", ], "gemini": [ + "gemini-3.8-flash", "gemini-3.7-flash", "gemini-3.1-pro-preview", "gemini-3-pro-preview", "gemini-3.6-flash", "gemini-3.1-flash-lite-preview", ], "zai": [ @@ -222,8 +223,8 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "gpt-5.3-codex-spark", "gpt-5.2", "gpt-5.2-codex", "gpt-5.1", "gpt-5.1-codex", "gpt-5.1-codex-max", "gpt-5.1-codex-mini", "gpt-5", "gpt-5-codex", "gpt-5-nano", "claude-fable-5", "claude-opus-5", "claude-sonnet-5", "claude-opus-4-8", "claude-opus-4-7", "claude-opus-4-6", "claude-opus-4-5", - "claude-sonnet-4-6", "claude-sonnet-4-5", "claude-sonnet-4", "claude-haiku-4-5", "gemini-3.7-flash", - "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro", "gemini-3-flash", + "claude-sonnet-4-6", "claude-sonnet-4-5", "claude-sonnet-4", "claude-haiku-4-5", "gemini-3.8-flash", + "gemini-3.7-flash", "gemini-3.6-flash", "gemini-3.5-flash", "gemini-3.5-flash-lite", "gemini-3.1-pro", "gemini-3-flash", "grok-4.6", "grok-4.5", "grok-build-0.1", "muse-spark-1.2", "minimax-m3", "minimax-m2.7", "minimax-m2.5", "glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5", "kimi-k2.7-code", "deepseek-v4-pro", "deepseek-v4-flash", "deepseek-v4-flash-free", "qwen3.6-plus", "qwen3.5-plus", "big-pickle", "mimo-v2.5-free", @@ -280,6 +281,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # only shows the configured model. IDs carry the "google/" publisher prefix Vertex expects # (see hermes_cli/model_setup_flows.py); validated live against a GCP project (global region). "vertex": [ + "google/gemini-3.8-flash", "google/gemini-3.7-flash", "google/gemini-3.1-pro-preview", "google/gemini-3-pro-preview", "google/gemini-3.6-flash", "google/gemini-3.5-flash", "google/gemini-3.5-flash-lite", "google/gemini-3-flash-preview", "google/gemini-3.1-flash-lite-preview", "google/gemini-3.1-flash-lite", diff --git a/plugins/model-providers/openrouter/__init__.py b/plugins/model-providers/openrouter/__init__.py index 6e82adfffd..a81ce3a12c 100644 --- a/plugins/model-providers/openrouter/__init__.py +++ b/plugins/model-providers/openrouter/__init__.py @@ -201,7 +201,7 @@ openrouter = OpenRouterProfile( base_url="https://openrouter.ai/api/v1", models_url="https://openrouter.ai/api/v1/models", fallback_models=( "anthropic/claude-sonnet-4.6", "openai/gpt-5.4", "deepseek/deepseek-chat", "google/gemini-3.8-flash", - "qwen/qwen3-plus", + "google/gemini-3.7-flash", "qwen/qwen3-plus", ), ) diff --git a/tests/agent/transports/test_chat_completions.py b/tests/agent/transports/test_chat_completions.py index 27f1767b9f..2531b10ea4 100644 --- a/tests/agent/transports/test_chat_completions.py +++ b/tests/agent/transports/test_chat_completions.py @@ -393,6 +393,20 @@ class TestChatCompletionsBuildKwargs: assert kw["max_tokens"] == GEMINI_DEFAULT_MAX_OUTPUT_TOKENS assert kw["extra_body"]["thinking_config"]["thinkingLevel"] == "high" + # Also verify gemini-3.8-flash gets headroom and correct thinking level + kw38 = transport.build_kwargs( + model="gemini-3.8-flash", + messages=[{"role": "user", "content": "Hi"}], + provider_profile=profile, + provider_name="gemini", + base_url=profile.base_url, + max_tokens=4096, + max_tokens_param_fn=lambda n: {"max_tokens": n}, + reasoning_config={"enabled": True, "effort": "medium"}, + ) + assert kw38["max_tokens"] == GEMINI_DEFAULT_MAX_OUTPUT_TOKENS + assert kw38["extra_body"]["thinking_config"]["thinkingLevel"] == "medium" + def test_gemini_without_thinking_keeps_explicit_max_tokens(self, transport): from providers import get_provider_profile From b706529476bf8ef843d6d03d785dcbc1dc8c5456 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:31:30 -0700 Subject: [PATCH 127/276] docs+test(models): Gemini 3.7/3.8 Flash in the Gemini and Vertex guides; curated Google Flash pickers must bill via the official snapshot Follow-up to the salvaged contributor commit: the Gemini and Vertex guide model tables now list both Flash generations the pickers offer, and an invariant test ties the OpenRouter/Nous curated google/gemini-*-flash entries to (a) a Google official-docs pricing row on the direct Gemini and Vertex routes and (b) membership in the direct gemini/vertex picker lists. Red on main (gemini-3.7-flash -> unknown), green with the contributor commit. Campaign: https://github.com/NousResearch/hermes-agent/issues/104154 --- tests/agent/test_usage_pricing.py | 28 ++++++++++++++++++++++++++++ website/docs/guides/google-gemini.md | 1 + website/docs/guides/google-vertex.md | 2 ++ 3 files changed, 31 insertions(+) diff --git a/tests/agent/test_usage_pricing.py b/tests/agent/test_usage_pricing.py index 4d4e8a1ea8..5895985e83 100644 --- a/tests/agent/test_usage_pricing.py +++ b/tests/agent/test_usage_pricing.py @@ -358,6 +358,34 @@ def test_vertex_default_model_estimates_cached_usage(monkeypatch): assert result.amount_usd is not None and result.amount_usd > 0 +def test_curated_google_flash_models_resolve_official_snapshot_pricing(monkeypatch): + """Every ``google/gemini-*-flash`` model curated for the OpenRouter and Nous + pickers must also bill through the Google official-docs snapshot on the + direct Gemini and Vertex routes — a model pickable via the aggregators but + ``unknown`` to Google-route accounting is a catalog/pricing drift. + """ + from hermes_cli.models_catalog_static import OPENROUTER_MODELS, _PROVIDER_MODELS + + monkeypatch.setattr( + "agent.usage_pricing.fetch_endpoint_model_metadata", + lambda *_args, **_kwargs: {}, + ) + curated = {m for m, _desc in OPENROUTER_MODELS} | set(_PROVIDER_MODELS["nous"]) + flash = sorted(m for m in curated if m.startswith("google/gemini-") and m.endswith("-flash")) + assert flash, "expected curated google/gemini-*-flash picker entries" + usage = CanonicalUsage(input_tokens=1_000_000, output_tokens=1_000_000, cache_read_tokens=1_000_000) + for model in flash: + bare = model.split("/", 1)[1] + gemini = estimate_usage_cost(bare, usage, provider="gemini") + vertex = estimate_usage_cost(model, usage, provider="vertex") + assert gemini.status == "estimated", (model, gemini.status) + assert gemini.source == "official_docs_snapshot", model + assert vertex.amount_usd == gemini.amount_usd, model + # Direct-route models the picker offers must also be pickable directly. + assert bare in _PROVIDER_MODELS["gemini"], model + assert model in _PROVIDER_MODELS["vertex"], model + + def test_normalize_usage_minimax_logs_cache_observability(caplog): """MiniMax providers on the Anthropic wire emit a debug-level cache-observability line recording the observable fields diff --git a/website/docs/guides/google-gemini.md b/website/docs/guides/google-gemini.md index c5d57fe40d..001c3cb710 100644 --- a/website/docs/guides/google-gemini.md +++ b/website/docs/guides/google-gemini.md @@ -104,6 +104,7 @@ The `hermes model` picker shows Gemini models maintained in Hermes' provider reg | Model | ID | Notes | |-------|----|-------| +| Gemini 3.8 Flash | `gemini-3.8-flash` | Most capable Flash model for long-horizon agentic and coding work | | Gemini 3.7 Flash | `gemini-3.7-flash` | Recommended default balance of speed, capability, and multimodal understanding | | Gemini 3.1 Pro Preview | `gemini-3.1-pro-preview` | Most capable reasoning, math, and coding model | | Gemini 3.5 Flash Lite | `gemini-3.5-flash-lite` | Fastest and lowest-cost option for lightweight tasks | diff --git a/website/docs/guides/google-vertex.md b/website/docs/guides/google-vertex.md index 54923db967..33bde3653b 100644 --- a/website/docs/guides/google-vertex.md +++ b/website/docs/guides/google-vertex.md @@ -88,6 +88,8 @@ Vertex requires the `google/` vendor prefix on model IDs. The `hermes model` pic | Model | ID | |-------|----| +| Gemini 3.8 Flash | `google/gemini-3.8-flash` | +| Gemini 3.7 Flash | `google/gemini-3.7-flash` | | Gemini 3.1 Pro Preview | `google/gemini-3.1-pro-preview` | | Gemini 3 Pro Preview | `google/gemini-3-pro-preview` | | Gemini 3 Flash Preview | `google/gemini-3-flash-preview` | From 54e24ae1fa09a172ccafb517719de07707185853 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 05:45:44 -0700 Subject: [PATCH 128/276] fix(nous): route anthropic/* over chat/completions by default (nous.anthropic_wire) Portal serves anthropic/* on two routes. The native /v1/messages wire, which Hermes has used since 02d5e230858, re-writes the previous turn's prompt cache on 14-20% of consecutive calls in concurrent tool loops; the chat/completions route does not. Measured 2026-09-06, 20 concurrent sessions x 6 tool calls on Fable 5.1, same account, same hour, same first-party pin: Nous /v1/messages 180 pairs, 25 stuck (13.9%); 3 earlier 40-session runs 15.0 / 17.3 / 20.2%; unchanged by the pin Nous /v1/chat/completions 320 pairs (2 runs), 0 stuck OpenRouter direct, pinned 161 pairs, 0 stuck "stuck" = cache_read on call k+1 equals cache_read on call k instead of call k's prompt size: the last write was not visible, the turn (~28K) was written twice. Every stuck pair had byte-identical system, tools and per-message shas, so it is not client-side mutation. At $10-20/M for cache writes that is 15-20% of a fan-out's write bill. The cause is inside the portal's native route (every element of its outgoing request reproduces clean from outside; NousResearch/api#227 carries the diagnostics). Until it is fixed, anthropic/* rides chat/completions. nous.anthropic_wire: native opts back in. Cost of chat: prior-turn thinking travels as OpenAI-style reasoning fields, and cache_control scopes are translated by the portal's adapter. Tests: new test_nous_anthropic_wire_default.py reads the knob through the real config loader in a temp HERMES_HOME and through resolve_runtime_provider (both settings); the existing native-wire contract suite selects native via an autouse fixture so the wire keeps working for the flip-back. Live: a real AIAgent on this branch, provider=nous + anthropic/claude-fable-5.1, dispatches to chat/completions with the session_id intact. --- hermes_cli/config_defaults.py | 7 +++ hermes_cli/providers.py | 27 ++++++++-- .../agent/test_nous_portal_anthropic_wire.py | 10 ++++ .../test_nous_anthropic_wire_default.py | 54 +++++++++++++++++++ website/docs/user-guide/configuring-models.md | 11 ++++ 5 files changed, 105 insertions(+), 4 deletions(-) create mode 100644 tests/hermes_cli/test_nous_anthropic_wire_default.py diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 1f98ab6024..12e08a5ecb 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -2284,6 +2284,13 @@ DEFAULT_CONFIG = { # server-issued credential lifetime (raising above it has no effect). 0 disables the # keepalive thread. "keepalive_interval_seconds": 900, + # anthropic_wire: which Portal route carries anthropic/* models. "chat" = + # /v1/chat/completions (default for now); "native" = /v1/messages, the Anthropic + # Messages wire (signed thinking passthrough, native cache_control scopes). Native is the + # better wire but re-writes the previous turn's cache on 14-20% of consecutive calls in + # concurrent tool loops (measured 2026-09-06; NousResearch/api#227), so chat is the + # default until that is fixed. + "anthropic_wire": "chat", }, # Google Vertex AI (Gemini). Auth is OAuth2 from a service-account JSON or ADC, NOT an API key; # the credential path lives in .env (VERTEX_CREDENTIALS_PATH / GOOGLE_APPLICATION_CREDENTIALS). diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index 20edd8837a..dbec5d698a 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -309,14 +309,33 @@ def host_mandated_api_mode(base_url: str = "") -> Optional[str]: def nous_api_mode(model: str = "") -> str: - """Wire protocol for a Nous Portal model: Portal serves its ``anthropic/*`` catalog on a native - Messages route alongside OpenAI-compatible chat/completions for everything else. Empty/unknown - model defaults to ``chat_completions`` (the historical Nous transport) as the safer path.""" + """Wire protocol for a Nous Portal model. Portal serves its ``anthropic/*`` catalog on a native + Messages route alongside OpenAI-compatible chat/completions for everything else. + + ``anthropic/*`` rides chat/completions by default for now (``nous.anthropic_wire``). Measured + 2026-09-06, 20 concurrent sessions x 6 tool calls on Fable 5.1, same account and hour: the + native route re-wrote the previous turn on 14-20% of consecutive calls (4 runs; the cache read + stopped at the prior breakpoint with byte-identical prefixes), chat/completions 0 of 320 pairs. + That is 15-20% of a fan-out's cache-write bill. The cause is inside the portal's native route + (NousResearch/api#227 carries the diagnostics); flip the default back to ``native`` when it is + fixed. Cost of ``chat``: prior-turn thinking travels as OpenAI-style reasoning fields instead of + signed native blocks, and cache_control scopes are translated by the portal's adapter. + Empty/unknown model defaults to ``chat_completions`` (the historical Nous transport).""" if str(model or "").strip().lower().startswith("anthropic/"): - return "anthropic_messages" + return "anthropic_messages" if _nous_anthropic_wire() == "native" else "chat_completions" return "chat_completions" +def _nous_anthropic_wire() -> str: + """``nous.anthropic_wire``: ``"chat"`` (default) or ``"native"``. Anything else reads as ``chat``.""" + try: + from hermes_cli.config import load_config_readonly + value = str(((load_config_readonly().get("nous") or {}).get("anthropic_wire")) or "chat").strip().lower() + except Exception: + return "chat" + return "native" if value == "native" else "chat" + + def determine_api_mode(provider: str, base_url: str = "", model: str = "") -> str: """API mode (wire protocol) for a provider/endpoint: host-mandated mode, then Nous dual-wire (model-derived — the overlay alone says openai_chat and would pin Claude on the wrong wire), diff --git a/tests/agent/test_nous_portal_anthropic_wire.py b/tests/agent/test_nous_portal_anthropic_wire.py index af85a6003e..32f9eb756e 100644 --- a/tests/agent/test_nous_portal_anthropic_wire.py +++ b/tests/agent/test_nous_portal_anthropic_wire.py @@ -21,8 +21,18 @@ from unittest.mock import MagicMock, patch import pytest from hermes_cli import runtime_provider as rp +from hermes_cli import providers as _providers from hermes_cli.providers import nous_api_mode + +@pytest.fixture(autouse=True) +def _native_wire_selected(monkeypatch): + """These contracts describe the native wire, which is now opt-in (``nous.anthropic_wire: + native``; default ``chat`` since 2026-09-06, see ``nous_api_mode``). Select it here so the + wire keeps working for the flip-back; the default's own contract is in + ``test_nous_anthropic_wire_default.py``.""" + monkeypatch.setattr(_providers, "_nous_anthropic_wire", lambda: "native") + PORTAL_URL = "https://inference-api.nousresearch.com/v1" # Staging / preview hosts used via NOUS_INFERENCE_BASE_URL — not the prod # hostname, so Portal behaviour must key off provider=nous. diff --git a/tests/hermes_cli/test_nous_anthropic_wire_default.py b/tests/hermes_cli/test_nous_anthropic_wire_default.py new file mode 100644 index 0000000000..907f6ade16 --- /dev/null +++ b/tests/hermes_cli/test_nous_anthropic_wire_default.py @@ -0,0 +1,54 @@ +"""``nous.anthropic_wire`` selects the Portal route for ``anthropic/*``: ``chat`` (default) rides +/v1/chat/completions, ``native`` rides /v1/messages. Read through the real config loader against a +temp HERMES_HOME (the loader is keyed on config path + mtime, so a fresh home is a fresh read), and +through ``resolve_runtime_provider`` so the api_mode a live agent gets is what is asserted.""" +from __future__ import annotations + +import pytest + +from hermes_cli import providers as _providers +from hermes_cli import runtime_provider as rp + +PORTAL = "https://inference-api.nousresearch.com/v1" + + +def _cfg(tmp_path, body: str, monkeypatch): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + (tmp_path / "config.yaml").write_text(body, encoding="utf-8") + + +def _portal_creds(monkeypatch): + monkeypatch.setattr(rp, "resolve_nous_runtime_credentials", + lambda **kw: {"base_url": PORTAL, "api_key": "jwt", "source": "portal", "expires_at": None}) + monkeypatch.setattr(rp, "_get_model_config", lambda: {"provider": "nous"}) + + +def test_default_is_chat_for_anthropic_and_unchanged_for_everything_else(tmp_path, monkeypatch): + _cfg(tmp_path, "model:\n default: anthropic/claude-fable-5.1\n", monkeypatch) + assert _providers.nous_api_mode("anthropic/claude-fable-5.1") == "chat_completions" + assert _providers.nous_api_mode("openai/gpt-5.6-sol") == "chat_completions" + assert _providers.determine_api_mode("nous", PORTAL, "anthropic/claude-fable-5.1") == "chat_completions" + + +def test_native_opt_in_restores_the_messages_wire_for_anthropic_only(tmp_path, monkeypatch): + _cfg(tmp_path, "nous:\n anthropic_wire: native\n", monkeypatch) + assert _providers.nous_api_mode("anthropic/claude-fable-5.1") == "anthropic_messages" + assert _providers.nous_api_mode("openai/gpt-5.6-sol") == "chat_completions" + + +@pytest.mark.parametrize("raw", ["''", "CHAT", "messages", "true", "1"]) +def test_anything_but_native_reads_as_chat(tmp_path, monkeypatch, raw): + _cfg(tmp_path, f"nous:\n anthropic_wire: {raw}\n", monkeypatch) + assert _providers.nous_api_mode("anthropic/claude-fable-5.1") == "chat_completions" + + +def test_runtime_resolution_hands_a_live_agent_the_selected_wire(tmp_path, monkeypatch): + """The path an AIAgent takes: provider=nous + anthropic model -> api_mode, both settings.""" + _portal_creds(monkeypatch) + _cfg(tmp_path, "model:\n provider: nous\n", monkeypatch) + resolved = rp.resolve_runtime_provider(requested="nous", target_model="anthropic/claude-fable-5.1") + assert (resolved["api_mode"], resolved["base_url"]) == ("chat_completions", PORTAL) + + _cfg(tmp_path, "model:\n provider: nous\nnous:\n anthropic_wire: native\n", monkeypatch) + resolved = rp.resolve_runtime_provider(requested="nous", target_model="anthropic/claude-fable-5.1") + assert resolved["api_mode"] == "anthropic_messages" diff --git a/website/docs/user-guide/configuring-models.md b/website/docs/user-guide/configuring-models.md index 0456c391e4..b9683f2826 100644 --- a/website/docs/user-guide/configuring-models.md +++ b/website/docs/user-guide/configuring-models.md @@ -234,6 +234,17 @@ omitted, Hermes keeps its normal provider and model capability detection. Older configs used a top-level `custom_providers:` list (with `base_url` instead of `api`). It still works and is auto-migrated to the `providers:` dict on `hermes update` (config v12). ::: +### Nous Portal: which wire carries Claude + +Nous Portal serves its `anthropic/*` models on two routes: OpenAI-compatible `/v1/chat/completions` and the native Anthropic Messages wire `/v1/messages`. `nous.anthropic_wire` picks one: + +```yaml +nous: + anthropic_wire: chat # default. "native" = the Anthropic Messages wire +``` + +`chat` is the default for now. The native wire is the better transport (signed thinking blocks pass through unchanged, native `cache_control` scopes), but in concurrent tool loops it currently re-writes the previous turn's prompt cache on 14–20% of consecutive calls, which is 15–20% of a fan-out's cache-write bill; the chat route measured 0 on the same test. Set `native` to opt back in (for example once the portal-side fix has shipped). Only `anthropic/*` models are affected; everything else on Nous already uses chat/completions. + ## When does it take effect? - **CLI** (`hermes chat`): next `hermes chat` invocation. From 190c73328c53b3024ab43afb701763fbeeccddd3 Mon Sep 17 00:00:00 2001 From: Xipong <217837358+Xipong@users.noreply.github.com> Date: Sat, 22 Aug 2026 18:19:32 +0300 Subject: [PATCH 129/276] chore(aux): carry local layer evolution onto current main --- agent/error_classifier.py | 3 ++- tests/agent/test_error_classifier.py | 17 +++++++++++++++++ 2 files changed, 19 insertions(+), 1 deletion(-) diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 177c5bbd84..ad4e9ae546 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -113,7 +113,8 @@ _BILLING_ERROR_CODES = frozenset({ # contains an overflow phrase; rate limit is matched first so throttle wins. _RATE_LIMIT_PATTERNS = ( "rate limit", "rate_limit", "too many requests", "throttled", "requests per minute", - "tokens per minute", "requests per day", "try again in", "please retry after", "resource_exhausted", + "tokens per minute", "requests per day", "try again in", "please retry after", + "resource exhausted", "resource_exhausted", "resource-exhausted", "resourceexhausted", "rate increased too quickly", "throttlingexception", "too many concurrent requests", "servicequotaexceededexception", "throttling", ) diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index b6156a9dc8..984607abd5 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -278,6 +278,23 @@ class TestClassifyApiError: assert result.reason == FailoverReason.rate_limit assert result.should_fallback is True + @pytest.mark.parametrize("spelling", [ + "resource exhausted", + "RESOURCE_EXHAUSTED", + "ResourceExhausted", + "resource-exhausted", + ]) + def test_resource_exhausted_separator_variants_without_status(self, spelling): + result = classify_api_error( + Exception(f"{spelling}: Worker local total request limit reached (32/32)"), + provider="nvidia", + model="nvidia/nemotron-3-ultra-550b-a55b", + ) + assert result.reason == FailoverReason.rate_limit + assert result.retryable is True + assert result.should_rotate_credential is True + assert result.should_fallback is True + def test_anthropic_429_usage_limit_without_reset_is_billing(self): e = MockAPIError( "usage limit reached", From 7166071fcaadb36df26f6d753dda97da6b5d699e Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:19:50 -0700 Subject: [PATCH 130/276] fix(auxiliary): ResourceExhausted spellings reach the payment/quota fallback gate The compression/aux fallback ladder decides "payment error" via _is_payment_error(), which only knew the spaced "resource exhausted". NVIDIA NIM and gRPC-style wrappers serialize the same quota signal as ResourceExhausted / RESOURCE_EXHAUSTED / resource-exhausted, so a 403/429/status-less body carrying it re-raised instead of walking the configured fallback_chain, and compression fell to the lossy emergency path. Add the three separator variants to _PAYMENT_KEYWORDS (same status gate, no broader "exhausted" matching). evals/auxiliary_resource_exhausted.py drives the real call_llm/async_call_llm against two local OpenAI-compatible listeners (nvidia profile primary, named custom fallback) for the before/after check. #85649 --- agent/auxiliary_client.py | 4 +- evals/auxiliary_resource_exhausted.py | 110 ++++++++++++++++++++++++++ tests/agent/test_auxiliary_client.py | 10 +++ 3 files changed, 123 insertions(+), 1 deletion(-) create mode 100644 evals/auxiliary_resource_exhausted.py diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index b62809ab4c..83f8586f76 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -2957,7 +2957,8 @@ def _contains_any(text: str, needles: Tuple[str, ...]) -> bool: # Billing-body markers (credit exhaustion wrapped in 402/403/404/429 bodies), plus daily/weekly quota -# exhaustion (functionally credit exhaustion; "resource exhausted" is the Vertex/gRPC quota phrasing). +# exhaustion (functionally credit exhaustion; "resource exhausted" is the Vertex/gRPC quota phrasing — +# also serialized by SDK wrappers and NIM as RESOURCE_EXHAUSTED / ResourceExhausted / resource-exhausted). _PAYMENT_KEYWORDS = ( "credits", "insufficient funds", "can only afford", "billing", "payment required", "out of funds", "run out of funds", "balance_depleted", "no usable credits", @@ -2965,6 +2966,7 @@ _PAYMENT_KEYWORDS = ( "requires a subscription", "upgrade for access", "upgrade for higher limits", "reached your session usage limit", "quota exceeded", "quota_exceeded", "too many tokens per day", "daily limit", "tokens per day", "daily quota", "resource exhausted", + "resource_exhausted", "resource-exhausted", "resourceexhausted", "weekly usage limit", "weekly limit", ) diff --git a/evals/auxiliary_resource_exhausted.py b/evals/auxiliary_resource_exhausted.py new file mode 100644 index 0000000000..2397d21a9d --- /dev/null +++ b/evals/auxiliary_resource_exhausted.py @@ -0,0 +1,110 @@ +"""Local HTTP contract probe; no vendor request or personal Hermes state. + +Run from the repository with its Python interpreter. JSON output identifies the +loaded module, observed SDK error, request order, and preserved message payload. +""" + +import argparse +import asyncio +import json +import os +from pathlib import Path +import sys +import tempfile +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--spelling", default="ResourceExhausted") + parser.add_argument("--mode", choices=("sync", "async", "sse"), default="sync") + parser.add_argument("--status", type=int, default=403) + args = parser.parse_args() + sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + requests = [] + message = f"{args.spelling}: Worker local total request limit reached (32/32)" + messages = [{"role": "user", "content": "Summarize: keep the deployment decision."}] + + class Handler(BaseHTTPRequestHandler): + def log_message(self, *args): + pass + + def do_POST(self): + body = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + if "messages" not in body: # capability probes (/api/show) are not chat requests + self.send_response(404) + self.end_headers() + return + requests.append(body) + primary = body["model"] == "primary-probe" + if primary and args.mode == "sse": + data = 'data: ' + json.dumps({"error": {"message": message}}) + '\n\ndata: [DONE]\n\n' + status, kind = 200, "text/event-stream" + elif primary: + data = json.dumps({"error": {"message": message, "type": "provider_error"}}) + status, kind = args.status, "application/json" + else: + data = json.dumps({"id": "local-summary", "object": "chat.completion", "created": 0, + "model": body["model"], "choices": [{"index": 0, "finish_reason": "stop", + "message": {"role": "assistant", "content": "Deployment decision preserved."}}]}) + status, kind = 200, "application/json" + encoded = data.encode() + self.send_response(status) + self.send_header("Content-Type", kind) + self.send_header("Content-Length", str(len(encoded))) + self.end_headers() + self.wfile.write(encoded) + + with tempfile.TemporaryDirectory(prefix="aux-resource-") as home: + os.environ["HOME"] = home + os.environ["HERMES_HOME"] = home + # Two listeners: a payment/quota error is credential-wide, so the fallback must be a + # different backend identity (distinct base_url) exactly as NIM -> OpenRouter is in the field. + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + fallback_server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + fallback_thread = threading.Thread(target=fallback_server.serve_forever, daemon=True) + thread.start() + fallback_thread.start() + base_url = f"http://127.0.0.1:{server.server_port}/v1" + fallback_url = f"http://127.0.0.1:{fallback_server.server_port}/v1" + # The primary is the real ``nvidia`` profile pointed at the local listener; the fallback is a + # named custom provider (its own credential label, like a second vendor in the field). + config = {"model": {"provider": "local-fallback", "model": "fallback-probe"}, + "providers": {"local-fallback": {"base_url": fallback_url, "api_key": "local-probe"}}, + "auxiliary": {"compression": {"provider": "nvidia", "model": "primary-probe", + "base_url": base_url, "api_key": "local-probe", "api_mode": "chat_completions", + "fallback_chain": [{"provider": "local-fallback", "model": "fallback-probe"}]}}} + # JSON is valid YAML and keeps this standalone probe dependency-free. + Path(home, "config.yaml").write_text(json.dumps(config), encoding="utf-8") + import logging + if os.environ.get("AUX_PROBE_DEBUG"): + logging.basicConfig(level=logging.DEBUG, stream=sys.stderr) + import agent.auxiliary_client as aux + + result = {"module": aux.__file__, "mode": args.mode, "status": args.status, + "spelling": args.spelling, "surface": "local HTTP contract, not live provider"} + try: + if args.mode == "async": + response = asyncio.run(aux.async_call_llm(task="compression", messages=messages, max_tokens=64)) + else: + with aux.aux_progress_hook((lambda: None) if args.mode == "sse" else None): + response = aux.call_llm(task="compression", messages=messages, max_tokens=64) + result["content"] = response.choices[0].message.content + except Exception as exc: + result["error"] = type(exc).__name__ + result["error_status"] = getattr(exc, "status_code", None) + result["error_message"] = str(exc) + finally: + for srv, thr in ((server, thread), (fallback_server, fallback_thread)): + srv.shutdown() + srv.server_close() + thr.join() + result["models"] = [request["model"] for request in requests] + result["messages_preserved"] = all(request["messages"] == messages for request in requests) + print(json.dumps(result)) + + +if __name__ == "__main__": + main() diff --git a/tests/agent/test_auxiliary_client.py b/tests/agent/test_auxiliary_client.py index 18c7c531b1..413f1ad339 100644 --- a/tests/agent/test_auxiliary_client.py +++ b/tests/agent/test_auxiliary_client.py @@ -1395,6 +1395,16 @@ class TestIsPaymentError: + @pytest.mark.parametrize("spelling", ["RESOURCE_EXHAUSTED", "ResourceExhausted", "resource-exhausted"]) + @pytest.mark.parametrize("status", [None, 429]) + def test_resource_exhausted_separator_variants_are_payment(self, spelling, status): + """NIM / gRPC wrappers serialize the quota signal without the space; the fallback gate + must read every spelling like the literal ``resource exhausted`` (#85649).""" + exc = Exception(f"{spelling}: Worker local total request limit reached (32/32)") + if status is not None: + exc.status_code = status + assert _is_payment_error(exc) is True + def test_403_subscription_required_is_payment(self): exc = Exception( "this model requires a subscription, upgrade for access: " From 335ecf9f4a0f056d170a661d33245c1769fd230a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 05:55:18 -0700 Subject: [PATCH 131/276] test: the two remaining native-wire contracts select nous.anthropic_wire=native explicitly test_nous_anthropic_fallback_uses_the_messages_wire and test_nous_child_rederives_api_mode_from_model describe the native wire, which is now opt-in; select it in the test the same way the wire-contract suite does, so both keep guarding the flip-back. --- tests/run_agent/test_provider_fallback.py | 9 ++++++--- tests/tools/test_delegate.py | 7 ++++++- 2 files changed, 12 insertions(+), 4 deletions(-) diff --git a/tests/run_agent/test_provider_fallback.py b/tests/run_agent/test_provider_fallback.py index 434ee306e6..8fffe2dda7 100644 --- a/tests/run_agent/test_provider_fallback.py +++ b/tests/run_agent/test_provider_fallback.py @@ -215,13 +215,16 @@ class TestFallbackChainAdvancement: assert mock_rpc.call_args.kwargs["explicit_api_key"] == "env-secret" - def test_nous_anthropic_fallback_uses_the_messages_wire(self): - """Portal Claude fallbacks must not stay on chat_completions. + def test_nous_anthropic_fallback_uses_the_messages_wire(self, monkeypatch): + """Portal Claude fallbacks must not stay on chat_completions when the native wire is selected. ``resolve_provider_client`` still returns an OpenAI client for Nous; activation has to re-derive api_mode from the model and rebuild the - Anthropic client — otherwise the turn POSTs /chat/completions. + Anthropic client — otherwise the turn POSTs /chat/completions. The wire + is opt-in since 2026-09-06 (``nous.anthropic_wire``, see ``nous_api_mode``). """ + from hermes_cli import providers as _providers + monkeypatch.setattr(_providers, "_nous_anthropic_wire", lambda: "native") portal = "https://inference-api.nousresearch.com/v1" fbs = [ { diff --git a/tests/tools/test_delegate.py b/tests/tools/test_delegate.py index 63d28cc121..58cb77fdc8 100644 --- a/tests/tools/test_delegate.py +++ b/tests/tools/test_delegate.py @@ -423,7 +423,12 @@ class TestDelegateTask(unittest.TestCase): def test_nous_child_rederives_api_mode_from_model(self): """Portal is dual-wire — same provider + different model prefix must - not inherit the parent's Messages/chat_completions mode verbatim.""" + not inherit the parent's Messages/chat_completions mode verbatim. + Native wire selected (opt-in since 2026-09-06, ``nous.anthropic_wire``).""" + with patch("hermes_cli.providers._nous_anthropic_wire", return_value="native"): + self._nous_child_rederives_api_mode_from_model() + + def _nous_child_rederives_api_mode_from_model(self): parent = _make_mock_parent(depth=0) parent.base_url = "https://inference-api.nousresearch.com/v1" parent.api_key = "portal-jwt" From 25a80b3bb6c1480c13639a03ed9714b5bca7646f Mon Sep 17 00:00:00 2001 From: Xipong <217837358+Xipong@users.noreply.github.com> Date: Sat, 5 Sep 2026 18:49:50 +0000 Subject: [PATCH 132/276] fix(skills): exclude generated runtime caches from bundled ownership hashes --- tests/tools/test_skills_runtime_cache.py | 161 +++++++++++++++++++++++ tools/skills_sync.py | 36 +++-- tools/skills_sync_bundled_ops.py | 10 +- tools/skills_sync_optional.py | 31 ++++- 4 files changed, 219 insertions(+), 19 deletions(-) create mode 100644 tests/tools/test_skills_runtime_cache.py diff --git a/tests/tools/test_skills_runtime_cache.py b/tests/tools/test_skills_runtime_cache.py new file mode 100644 index 0000000000..dc4dce8f0c --- /dev/null +++ b/tests/tools/test_skills_runtime_cache.py @@ -0,0 +1,161 @@ +"""Runtime caches must not take ownership of updater-managed skill packages.""" + +import hashlib +import py_compile + +import pytest + +from tools import skills_sync as ss +from tools.skills_sync_bundled_ops import diff_bundled_skill, list_user_modified_bundled_skills +from tools.skills_sync_optional import _skill_file_list + + +def _legacy_hash(directory): + digest = hashlib.md5() + for path in sorted(directory.rglob("*")): + if path.is_file(): + digest.update(str(path.relative_to(directory)).encode("utf-8")) + digest.update(path.read_bytes()) + return digest.hexdigest() + + +def _write(root, rel, text): + target = root / rel + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(text, encoding="utf-8") + return target + + +@pytest.fixture +def skill_tree(tmp_path, monkeypatch): + base = tmp_path / "profile" + bundled = base / "bundled" + src = bundled / "coding" / "demo" + _write(src, "SKILL.md", "---\nname: demo\n---\n# Demo\n") + _write(src, "scripts/helper.py", "ANSWER = 1\n") + skills = base / "skills" + monkeypatch.setattr(ss, "HERMES_HOME", base) + monkeypatch.setattr(ss, "SKILLS_DIR", skills) + monkeypatch.setattr(ss, "MANIFEST_FILE", skills / ".bundled_manifest") + monkeypatch.setattr(ss, "_get_bundled_dir", lambda: bundled) + monkeypatch.setattr(ss, "_get_optional_dir", lambda: base / "no-optional") + monkeypatch.setattr(ss, "_build_external_skill_index", set) + monkeypatch.setattr(ss, "_read_suppressed_names", set) + assert ss.sync_skills(quiet=True)["copied"] == ["demo"] + return src, skills / "coding" / "demo" + + +def test_real_python_compilation_does_not_freeze_update(skill_tree): + src, dest = skill_tree + origin = ss._read_manifest()["demo"] + py_compile.compile(str(dest / "scripts/helper.py"), doraise=True) + assert list((dest / "scripts/__pycache__").glob("*.pyc")) + assert ss._dir_hash(dest) == origin + assert list_user_modified_bundled_skills() == [] + assert diff_bundled_skill("demo")["diffs"] == [] + _write(src, "scripts/helper.py", "ANSWER = 2\n") + result = ss.sync_skills(quiet=True) + assert result["updated"] == ["demo"] + assert (dest / "scripts/helper.py").read_text() == "ANSWER = 2\n" + assert ss._read_manifest()["demo"] == ss._dir_hash(src) + + +@pytest.mark.parametrize("cache_path", [ + "scripts/__pycache__/helper.cpython-311.pyc", + ".pytest_cache/v/cache/nodeids", + "scripts/.mypy_cache/3.11/helper.data.json", + "scripts/.ruff_cache/0.15.0/abc", + "scripts/helper.pyc", "scripts/helper.pyo", +]) +def test_runtime_cache_is_not_hash_or_diff_content(skill_tree, cache_path): + src, dest = skill_tree + _write(dest, cache_path, "generated") + assert ss._dir_hash(dest) == ss._dir_hash(src) + assert cache_path not in _skill_file_list(dest) + assert diff_bundled_skill("demo")["modified"] is False + + +@pytest.mark.parametrize("edited_path", ["SKILL.md", "scripts/helper.py", "references/notes.md", "cache/data.json", "scripts/bytecode_only.pyc"]) +def test_genuine_edits_next_to_cache_are_preserved(skill_tree, edited_path): + src, dest = skill_tree + _write(dest, "scripts/__pycache__/helper.cpython-311.pyc", "generated") + _write(dest, edited_path, "user-owned content") + _write(src, "references/new-upstream.md", "new upstream documentation") + assert [entry["name"] for entry in list_user_modified_bundled_skills()] == ["demo"] + result = ss.sync_skills(quiet=True) + assert result["user_modified"] == ["demo"] + assert (dest / edited_path).read_text() == "user-owned content" + assert not (dest / "references/new-upstream.md").exists() + + +def test_source_cache_is_neither_seeded_nor_recorded(skill_tree): + src, dest = skill_tree + _write(src, "scripts/__pycache__/helper.cpython-311.pyc", "source generated") + _write(src, "SKILL.md", "---\nname: demo\n---\n# Changed upstream\n") + assert ss.sync_skills(quiet=True)["updated"] == ["demo"] + assert not (dest / "scripts/__pycache__").exists() + ss._rmtree_writable(dest) + ss._write_manifest({}) + assert ss.sync_skills(quiet=True)["copied"] == ["demo"] + assert not (dest / "scripts/__pycache__").exists() + + +def test_legacy_cache_hash_migrates_only_with_matching_origin(skill_tree): + src, dest = skill_tree + _write(dest, "scripts/__pycache__/helper.cpython-311.pyc", "legacy generated") + ss._write_manifest({"demo": _legacy_hash(dest)}) + assert list_user_modified_bundled_skills() == [] + _write(src, "scripts/helper.py", "ANSWER = 2\n") + assert ss.sync_skills(quiet=True)["updated"] == ["demo"] + assert ss._read_manifest()["demo"] == ss._dir_hash(src) + + +def test_legacy_hash_mismatch_never_rebaselines_user_edits(skill_tree): + src, dest = skill_tree + cache = _write(dest, "scripts/__pycache__/helper.cpython-311.pyc", "legacy generated") + origin = _legacy_hash(dest) + ss._write_manifest({"demo": origin}) + cache.unlink() + _write(dest, "scripts/helper.py", "user edit\n") + _write(src, "scripts/helper.py", "upstream edit\n") + assert ss.sync_skills(quiet=True)["user_modified"] == ["demo"] + assert ss._read_manifest()["demo"] == origin + assert (dest / "scripts/helper.py").read_text() == "user edit\n" + + +def test_clean_manifest_hash_is_backwards_compatible(skill_tree): + src, dest = skill_tree + assert ss._dir_hash(src) == _legacy_hash(src) + assert ss._dir_hash(dest) == _legacy_hash(dest) + + +def test_runtime_cache_does_not_hide_dotfile_edit(skill_tree): + src, dest = skill_tree + _write(dest, ".settings.json", "user settings") + _write(dest, ".pytest_cache/v/cache/nodeids", "[]") + assert ss._dir_hash(dest) != ss._dir_hash(src) + assert ".settings.json" in _skill_file_list(dest) + assert [e["path"] for e in diff_bundled_skill("demo")["diffs"]] == [".settings.json"] + + +def test_legacy_cache_origin_allows_rename_without_freezing(skill_tree): + src, dest = skill_tree + _write(dest, "scripts/__pycache__/helper.cpython-311.pyc", "legacy generated") + ss._write_manifest({"demo": _legacy_hash(dest)}) + new_src = src.parent.parent / "recategorized" / "demo" + new_src.parent.mkdir(parents=True) + src.rename(new_src) + _write(new_src, "scripts/helper.py", "ANSWER = 3\n") + result = ss.sync_skills(quiet=True) + assert result["updated"] == ["demo"] + assert not dest.exists() + assert (ss._skills_dir() / "recategorized/demo/scripts/helper.py").read_text() == "ANSWER = 3\n" + + +def test_hash_filter_is_skill_relative(tmp_path): + # A Python environment/install prefix can itself contain a cache name. + src = tmp_path / "__pycache__" / "demo" + _write(src, "SKILL.md", "real skill content") + _write(src, "scripts/helper.py", "ANSWER = 1\n") + assert ss._dir_hash(src) == _legacy_hash(src) + assert set(_skill_file_list(src)) == {"SKILL.md", "scripts/helper.py"} diff --git a/tools/skills_sync.py b/tools/skills_sync.py index 322167e5ad..ca774820be 100644 --- a/tools/skills_sync.py +++ b/tools/skills_sync.py @@ -25,8 +25,9 @@ for _stream in (sys.stdout, sys.stderr): from hermes_constants import get_bundled_skills_dir, get_hermes_home, get_optional_skills_dir from agent.skill_utils import ESSENTIAL_SKILLS, is_excluded_skill_path from tools.skill_usage import _read_skill_name, read_suppressed_names -from tools.skills_sync_bundled_ops import _is_tracked_user_modification -from tools.skills_sync_optional import _backfill_optional_provenance, _read_hub_install_paths +from tools.skills_sync_optional import ( + _backfill_optional_provenance, _ignore_runtime_cache, _is_runtime_cache, _read_hub_install_paths, +) from utils import atomic_write_text logger = logging.getLogger(__name__) @@ -150,17 +151,34 @@ def _compute_relative_dest(skill_dir: Path, bundled_dir: Path) -> Path: return _skills_dir() / skill_dir.relative_to(bundled_dir) -def _dir_hash(directory: Path) -> str: - """MD5 over relative paths + contents of every file in a directory.""" +def _dir_hash(directory: Path, *, include_runtime_cache: bool = False) -> str: + """MD5 of package paths/content, excluding generated runtime state. + + The legacy option is only for proving an exact pre-filter origin match. + Keep the original path encoding so clean existing manifests remain valid. + """ hasher = hashlib.md5() with suppress(OSError): for fpath in sorted(directory.rglob("*")): - if fpath.is_file(): + if (include_runtime_cache or not _is_runtime_cache(fpath, directory)) and fpath.is_file(): hasher.update(str(fpath.relative_to(directory)).encode("utf-8")) hasher.update(fpath.read_bytes()) return hasher.hexdigest() +def _matches_origin_hash(directory: Path, origin_hash: str, user_hash: Optional[str] = None) -> bool: + """Prove unchanged package ownership against a clean OR exact legacy hash. + + Never re-baseline a differing package merely because it contains a cache: + if legacy cached bytes changed/disappeared, the old origin cannot be proven. + A genuinely edited package must remain protected in that case. + """ + if not origin_hash: + return False + current = _dir_hash(directory) if user_hash is None else user_hash + return current == origin_hash or _dir_hash(directory, include_runtime_cache=True) == origin_hash + + def _move_dir(src: Path, dest: Path) -> None: dest.parent.mkdir(parents=True, exist_ok=True) shutil.move(str(src), str(dest)) @@ -168,7 +186,7 @@ def _move_dir(src: Path, dest: Path) -> None: def _copy_dir(src: Path, dest: Path) -> None: dest.parent.mkdir(parents=True, exist_ok=True) - shutil.copytree(src, dest) + shutil.copytree(src, dest, ignore=_ignore_runtime_cache) def _recover_renamed_skill(st: "_SyncState", skill_name: str, dest: Path) -> Optional[str]: @@ -192,7 +210,7 @@ def _recover_renamed_skill(st: "_SyncState", skill_name: str, dest: Path) -> Opt continue if rel in st.hub_paths: # the hub owns its install paths continue - if _dir_hash(candidate) != origin_hash: # moving a customized copy would edit user work + if not _matches_origin_hash(candidate, origin_hash): # moving a customized copy would edit user work st.say( f" ⚠ {skill_name}: upstream moved this skill to {_rel_skills_posix(dest)}, but your " f"modified copy at {rel} was kept — it will not receive updates. " @@ -284,7 +302,7 @@ def _replace_skill_dir(skill_src: Path, dest: Path) -> None: _rmtree_writable(backup) shutil.move(str(dest), str(backup)) try: - shutil.copytree(skill_src, dest) + shutil.copytree(skill_src, dest, ignore=_ignore_runtime_cache) except OSError: if backup.exists(): # clear a partially-written dest so it can't shadow/block the restore if dest.exists(): @@ -313,7 +331,7 @@ def _update_existing_skill(st: _SyncState, skill_name: str, skill_src: Path, des st.manifest[skill_name] = user_hash st.skipped += 1 return - if _is_tracked_user_modification(origin_hash, user_hash): + if not _matches_origin_hash(dest, origin_hash, user_hash): st.user_modified.append(skill_name) st.say(f" ~ {skill_name} (user-modified, skipping)") return diff --git a/tools/skills_sync_bundled_ops.py b/tools/skills_sync_bundled_ops.py index fe63ec96ec..5f66e12025 100644 --- a/tools/skills_sync_bundled_ops.py +++ b/tools/skills_sync_bundled_ops.py @@ -7,12 +7,6 @@ from typing import List, Optional, Tuple from tools.skills_sync_optional import _skill_file_list, _ss -def _is_tracked_user_modification(origin_hash: str, user_hash: str) -> bool: - """User modification ``hermes update`` keeps: a recorded origin hash (un-baselined v1 entries - don't count) AND differing content. Shared by the sync loop and list-modified (no drift).""" - return bool(origin_hash) and user_hash != origin_hash - - def _bundled_state(): """``(skills_sync, manifest, bundled_dir, {skill_name: bundled_src})`` — shared op preamble.""" ss = _ss() @@ -75,7 +69,7 @@ def list_user_modified_bundled_skills() -> List[dict]: for skill_name, skill_dir in ss._discover_bundled_skills(bundled_dir): origin_hash = manifest.get(skill_name, "") # empty = untracked/un-baselined v1: next sync handles it dest = ss._compute_relative_dest(skill_dir, bundled_dir) - if origin_hash and dest.exists() and _is_tracked_user_modification(origin_hash, ss._dir_hash(dest)): + if origin_hash and dest.exists() and not ss._matches_origin_hash(dest, origin_hash): modified.append({"name": skill_name, "dest": dest, "bundled_src": skill_dir}) return sorted(modified, key=lambda e: e["name"]) @@ -180,7 +174,7 @@ def remove_pristine_bundled_skills(dry_run: bool = False) -> dict: if not dry_run: # already gone from disk; forget the stale manifest entry manifest.pop(name, None) continue - if ss._dir_hash(dest) != origin_hash: + if not ss._matches_origin_hash(dest, origin_hash): skipped.append({"name": name, "reason": "user-modified (kept)"}) continue if not dry_run: diff --git a/tools/skills_sync_optional.py b/tools/skills_sync_optional.py index 20bcf96936..313de12c88 100644 --- a/tools/skills_sync_optional.py +++ b/tools/skills_sync_optional.py @@ -38,9 +38,36 @@ def _safe_rel_install_path(path: Path, base: Path) -> str: return "/".join(parts) +# Only generated runtime state, never generic cache/ or arbitrary dotfiles. This +# is updater ownership, not the security scanner's content-integrity policy. +_RUNTIME_CACHE_DIRS = frozenset({"__pycache__", ".pytest_cache", ".mypy_cache", ".ruff_cache"}) + + +def _is_runtime_cache(path: Path, skill_dir: Path) -> bool: + """Whether a skill-relative path is disposable Python/tool runtime state. + + Install prefixes may themselves contain a cache directory name. Legacy + sibling bytecode is ignored only alongside its source; source-less .pyc + files can be deliberately shipped or user-owned content. + """ + relative = path.relative_to(skill_dir) + if any(part in _RUNTIME_CACHE_DIRS for part in relative.parts[:-1]): + return True + if path.name in _RUNTIME_CACHE_DIRS and path.is_dir(): + return True + return path.suffix in {".pyc", ".pyo"} and path.with_suffix(".py").is_file() + + +def _ignore_runtime_cache(directory: str, names: List[str]) -> List[str]: + """copytree callback: don't seed generated caches into managed packages.""" + root = Path(directory) + return [name for name in names if _is_runtime_cache(root / name, root)] + + def _skill_file_list(skill_dir: Path) -> List[str]: - """List files inside a skill directory in lock-file format.""" - return [f.relative_to(skill_dir).as_posix() for f in sorted(skill_dir.rglob("*")) if f.is_file()] + """List package files for provenance/diff, excluding generated runtime caches.""" + return [f.relative_to(skill_dir).as_posix() for f in sorted(skill_dir.rglob("*")) + if not _is_runtime_cache(f, skill_dir) and f.is_file()] def _load_hub_lock() -> Optional[dict]: From 6ced334f550229c498388c5f60e0c45fb129a6a1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:27:43 -0700 Subject: [PATCH 133/276] docs(skills): note that runtime caches never count as bundled-skill edits --- website/docs/user-guide/features/skills.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/website/docs/user-guide/features/skills.md b/website/docs/user-guide/features/skills.md index 6246190fc2..cf856f4495 100644 --- a/website/docs/user-guide/features/skills.md +++ b/website/docs/user-guide/features/skills.md @@ -1009,6 +1009,8 @@ On each sync, Hermes recomputes the hash of your local copy and compares it to t - **Unchanged** → safe to pull upstream changes, copy the new bundled version in, record the new origin hash. - **Changed** → treated as **user-modified** and skipped forever, so your edits never get stomped. +Generated runtime caches inside a skill (`__pycache__/`, `.pytest_cache/`, `.mypy_cache/`, `.ruff_cache/`, and a `.pyc` sitting next to its `.py`) are not part of the hash, so running a skill's helper script never marks it user-modified or hides it from `hermes skills list-modified` / `diff`. + The protection is good, but it has one sharp edge. If you edit a bundled skill and then later want to abandon your changes and go back to the bundled version by just copy-pasting from `~/.hermes/hermes-agent/skills/`, the manifest still holds the *old* origin hash from whenever the last successful sync ran. Your fresh copy-paste contents (current bundled hash) won't match that stale origin hash, so sync keeps flagging it as user-modified. `hermes skills reset` is the escape hatch: From 028fe2c4c857710ab335a8455f5cbbcc52cbf795 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 06:19:11 -0700 Subject: [PATCH 134/276] =?UTF-8?q?feat(delegation):=20per-task=20completi?= =?UTF-8?q?on=20groups=20=E2=80=94=20ungrouped=20subagents=20return=20as?= =?UTF-8?q?=20they=20finish?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A background delegate_task call used to be ONE async unit: the runner joined on every child and a single consolidated message re-entered the conversation when the SLOWEST finished. Fifteen independent PR reviews therefore waited on the fifteenth before the parent could act on the first. Each task now carries an optional `group`. `_units_of` partitions the call's children into units — one per distinct group, one per ungrouped task — and each unit is dispatched to the async registry on its own, so its results re-enter the conversation as soon as THAT unit is done. Tasks that must be compared or merged share a group and still return together. Capacity is unchanged: every unit of one call joins the first unit's pool slot (`slot_key` in `async_delegation._dispatch`), so splitting never consumes more of `delegation.max_concurrent_children` than the call did. Unit ids suffix the call's id (`deleg_xxxx-1`, `-2`, …) so live transcripts stay under one dir; the completion block names the group and notes that sibling units report separately; `active_task_count` counts a unit's own tasks. --- tests/tools/test_async_delegation.py | 84 +++++++++ tools/AGENTS.md | 5 +- tools/async_delegation.py | 43 +++-- tools/delegate_tool.py | 12 +- tools/delegate_tool_dispatch.py | 163 +++++++++++------- tools/process_registry_notifications.py | 12 +- .../docs/user-guide/features/delegation.md | 18 +- 7 files changed, 254 insertions(+), 83 deletions(-) diff --git a/tests/tools/test_async_delegation.py b/tests/tools/test_async_delegation.py index 54e9943179..61d3e3899e 100644 --- a/tests/tools/test_async_delegation.py +++ b/tests/tools/test_async_delegation.py @@ -946,3 +946,87 @@ def test_batch_model_rejection_notice_requires_configured_model_in_text(monkeypa text = format_process_notification(evt) assert text is not None assert "SUBAGENT MODEL REJECTED" not in text + + +# --------------------------------------------------------------------------- +# Per-group completion units: ungrouped tasks return alone, a `group` returns +# together, and the units of one call share ONE capacity slot. +# --------------------------------------------------------------------------- + +def _grouped_fanout(monkeypatch, tasks, gates): + """delegate_task(tasks) in the background with gated fake children; returns the parsed handle.""" + from unittest.mock import MagicMock + import tools.delegate_tool as dt + + parent = MagicMock() + parent._delegate_depth = 0 + parent.session_id = "sess" + parent._interrupt_requested = False + parent._active_children = [] + parent._active_children_lock = None + + def child(task_index, goal, child=None, parent_agent=None, **kw): + gates[task_index].wait(timeout=60) + return {"task_index": task_index, "status": "completed", "summary": f"done: {goal}", "api_calls": 1, + "duration_seconds": 0.1, "model": "m", "exit_reason": "completed"} + + def build(**kw): + c = MagicMock() + c._delegate_role = "leaf" + c._subagent_id = f"s{kw['task_index']}" + return c + + creds = {"model": "m", "provider": None, "base_url": None, "api_key": None, "api_mode": None, "command": None, + "args": None} + monkeypatch.setattr(dt, "_build_child_agent", build) + monkeypatch.setattr(dt, "_run_single_child", child) + monkeypatch.setattr(dt, "_resolve_delegation_credentials", lambda *a, **k: creds) + return json.loads(dt.delegate_task(tasks=tasks, background=True, parent_agent=parent)) + + +def test_ungrouped_task_completes_alone_and_group_completes_together(monkeypatch): + """A finished ungrouped task must not wait for its siblings; tasks sharing a `group` must.""" + gates = [threading.Event() for _ in range(4)] + tasks = [ + {"goal": "review PR 1 thoroughly and report"}, + {"goal": "compare approach A in detail", "group": "cmp"}, + {"goal": "compare approach B in detail", "group": "cmp"}, + {"goal": "review PR 2 thoroughly and report"}, + ] + handle = _grouped_fanout(monkeypatch, tasks, gates) + assert handle["status"] == "dispatched" + by_group = {tuple(u["task_indexes"]): u["group"] for u in handle["units"]} + assert by_group == {(0,): None, (1, 2): "cmp", (3,): None} + + gates[3].set() + evt = _drain_one() + assert [r["task_index"] for r in evt["results"]] == [3] # PR 2 landed while everything else still runs + assert "review PR 2" in format_process_notification(evt) + + gates[1].set() + assert _drain_one(timeout=0.5) is None # half a group is not a completion + gates[2].set() + evt = _drain_one() + assert evt["group"] == "cmp" and [r["task_index"] for r in evt["results"]] == [1, 2] + + gates[0].set() + assert [r["task_index"] for r in _drain_one()["results"]] == [0] + + +def test_units_of_one_call_share_a_single_capacity_slot(): + """Splitting a call into per-task completions must not consume more pool capacity than the call did.""" + gate = threading.Event() + + def blocker(): + gate.wait(timeout=60) + return {"results": [], "total_duration_seconds": 0} + + common = dict(goals=["a", "b"], context=None, toolsets=None, role="leaf", model="m", session_key="", + runner=blocker, max_async_children=1) + first = ad.dispatch_async_delegation_batch(delegation_id="deleg_call-1", task_indexes=[0], **common) + second = ad.dispatch_async_delegation_batch(delegation_id="deleg_call-2", task_indexes=[1], + slot_key="deleg_call-1", **common) + other = ad.dispatch_async_delegation_batch(delegation_id="deleg_other", **common) + assert (first["status"], second["status"], other["status"]) == ("dispatched", "dispatched", "rejected") + assert ad.active_task_count() == 2 + gate.set() diff --git a/tools/AGENTS.md b/tools/AGENTS.md index 7e6526a72a..fa624f2bb7 100644 --- a/tools/AGENTS.md +++ b/tools/AGENTS.md @@ -79,7 +79,10 @@ fixed at the mount, not by adding a tool. Spawns a subagent with isolated context + terminal session; the parent waits for the summary unless `background=true`, which returns a delegation id and re-enters the result via the async-delegation completion queue. Shapes: single (`goal` + optional `context`, `toolsets`) or batch (`tasks: [...]`, -concurrency capped by `delegation.max_concurrent_children`, default 3). Roles: `leaf` (default; +concurrency capped by `delegation.max_concurrent_children`, default 3). A background batch is split +into completion **units** (`delegate_tool_dispatch._units_of`): tasks sharing a `group` join and +report together; each ungrouped task reports alone as it finishes. Units of one call share ONE pool +slot (`slot_key` in `async_delegation._dispatch`) — never count units against capacity. Roles: `leaf` (default; no `delegate_task`, `clarify`, `memory`, `send_message`, `cronjob`; keeps `execute_code`) and `orchestrator` (keeps `delegate_task`; gated by `delegation.orchestrator_enabled`, bounded by `delegation.max_spawn_depth`, default 2). Config knobs under `delegation:`: diff --git a/tools/async_delegation.py b/tools/async_delegation.py index 1ef7fbdc27..9efa18af4f 100644 --- a/tools/async_delegation.py +++ b/tools/async_delegation.py @@ -415,7 +415,7 @@ def _get_executor(max_workers: int) -> ThreadPoolExecutor: def active_count() -> int: - """Number of live async delegation UNITS (a whole batch counts as ONE slot).""" + """Number of live async delegation UNITS (one per completion message: a task group or an ungrouped task).""" with _records_lock: return sum(1 for r in _records.values() if r.get("status") in _LIVE_STATES) @@ -425,7 +425,8 @@ def active_task_count() -> int: no goal list counts 1) — the truthful observability figure, unlike slots.""" with _records_lock: return sum( - len(r["goals"]) if r.get("is_batch") and isinstance(r.get("goals"), (list, tuple)) and r["goals"] else 1 + len(r.get("task_indexes") or r["goals"]) + if r.get("is_batch") and isinstance(r.get("goals"), (list, tuple)) and r["goals"] else 1 for r in _records.values() if r.get("status") in {"running", "finalizing"}) @@ -496,12 +497,15 @@ def _dispatch( toolsets: Optional[List[str]], role: str, model: Optional[str], session_key: str, parent_session_id: Optional[str], runner: Callable[[], Dict[str, Any]], origin_ui_session_id: str, origin_session_id: str, interrupt_fn: Optional[Callable[[], None]], max_async_children: int, - progress_fn: Optional[Callable[[], tuple]], capacity_error: str, + progress_fn: Optional[Callable[[], tuple]], capacity_error: str, slot_key: Optional[str] = None, + task_indexes: Optional[List[int]] = None, ) -> Dict[str, Any]: """Shared dispatch core for single (``goals is None``) and batch units. Capacity check + record insert happen under ONE lock hold so concurrent dispatches can't both pass the check and exceed the cap. At capacity the dispatch is REJECTED (never queued) so a runaway model - can't pile up unbounded background work.""" + can't pile up unbounded background work. ``slot_key`` names the pool slot the unit occupies + (default: its own id); the units of one delegate_task call share the first unit's id so + splitting a call into per-group completions never consumes more capacity than the call did.""" is_batch = goals is not None label = " batch" if is_batch else "" classify = _batch_status if is_batch else (lambda r: r.get("status") or "completed") @@ -515,11 +519,14 @@ def _dispatch( **_capture_routing_origin(), "status": "running", "dispatched_at": dispatched_at, "completed_at": None, "interrupt_fn": interrupt_fn, **({"is_batch": True} if is_batch else {}), "progress_fn": progress_fn, + "slot_key": slot_key or delegation_id, + # Which of the call's ``goals`` this unit runs (None = all of them). + **({"task_indexes": list(task_indexes)} if task_indexes is not None else {}), # Stale-monitor bookkeeping (see _stale_monitor_loop). "_progress_token": None, "_progress_ts": dispatched_at, "_interrupted_at": None} with _records_lock: - running = sum(1 for r in _records.values() if r.get("status") in _ACTIVE_STATES) - if running >= max_async_children: + active_slots = {r.get("slot_key") or r["delegation_id"] for r in _records.values() if r.get("status") in _ACTIVE_STATES} + if record["slot_key"] not in active_slots and len(active_slots) >= max_async_children: return {"status": "rejected", "error": capacity_error} _records[delegation_id] = record _persist_dispatch(record) @@ -584,21 +591,26 @@ def dispatch_async_delegation_batch( session_key: str, parent_session_id: Optional[str] = None, runner: Callable[[], Dict[str, Any]], origin_ui_session_id: str = "", origin_session_id: str = "", interrupt_fn: Optional[Callable[[], None]] = None, max_async_children: int = _DEFAULT_MAX_ASYNC_CHILDREN, delegation_id: Optional[str] = None, - progress_fn: Optional[Callable[[], tuple]] = None, + progress_fn: Optional[Callable[[], tuple]] = None, slot_key: Optional[str] = None, + task_indexes: Optional[List[int]] = None, ) -> Dict[str, Any]: - """Dispatch a WHOLE fan-out batch as ONE background unit: ``runner`` runs the - entire batch and returns the combined ``{"results": [...], "total_duration_seconds": N}`` - dict. The batch occupies ONE async slot (in-batch parallelism is bounded - separately) and produces a SINGLE completion event carrying per-task ``results``.""" + """Dispatch a fan-out unit (a whole batch, or one ``group`` of a delegate_task call) as ONE + background unit: ``runner`` runs its tasks and returns the combined ``{"results": [...], + "total_duration_seconds": N}`` dict. The unit occupies ONE async slot — or joins the slot named + by ``slot_key`` (in-unit parallelism is bounded separately) — and produces a SINGLE completion + event carrying per-task ``results``.""" delegation_id = delegation_id or _new_delegation_id() - n = len(goals) - combined_goal = goals[0] if n == 1 else f"{n} parallel subagents: " + "; ".join(g[:40] for g in goals) + # ``goals`` is the whole call (result task_index indexes it); the unit's own goals label the record. + unit_goals = [goals[i] for i in task_indexes] if task_indexes is not None else list(goals) + n = len(unit_goals) + combined_goal = unit_goals[0] if n == 1 else f"{n} parallel subagents: " + "; ".join(g[:40] for g in unit_goals) handle = _dispatch( delegation_id=delegation_id, goal=combined_goal, goals=goals, context=context, toolsets=toolsets, role=role, model=model, session_key=session_key, parent_session_id=parent_session_id, runner=runner, origin_ui_session_id=origin_ui_session_id, origin_session_id=origin_session_id, - interrupt_fn=interrupt_fn, max_async_children=max_async_children, progress_fn=progress_fn, + interrupt_fn=interrupt_fn, max_async_children=max_async_children, progress_fn=progress_fn, slot_key=slot_key, + task_indexes=task_indexes, capacity_error=( f"Async delegation capacity reached ({max_async_children} running). Wait for one to finish " "(its result will re-enter the chat), or raise delegation.max_concurrent_children in " @@ -651,7 +663,8 @@ def _push_completion_event(record: Dict[str, Any], result: Dict[str, Any], statu payload = { "is_batch": True, "results": result.get("results") or [], "live_transcripts": result.get("live_transcripts"), "error": result.get("error"), - "total_duration_seconds": result.get("total_duration_seconds")} + "total_duration_seconds": result.get("total_duration_seconds"), + **({"group": result["group"]} if result.get("group") is not None else {})} else: payload = { "summary": result.get("summary"), "error": result.get("error"), "api_calls": result.get("api_calls", 0), diff --git a/tools/delegate_tool.py b/tools/delegate_tool.py index 70e98fe6a6..5845be46ab 100644 --- a/tools/delegate_tool.py +++ b/tools/delegate_tool.py @@ -457,8 +457,9 @@ _DESCRIPTION_HEAD = ( "Spawn subagents in isolated contexts; each gets its own conversation, terminal session, and toolset, and only its " "final summary returns to you. Pass every task in `tasks` — one entry spawns one subagent, several run in parallel " "(limit in the tasks description).\n\n" - "Runs in the background: dispatch returns immediately with live transcript paths, and the completed result (one " - "consolidated message, results in task order) re-enters the conversation on its own. Do NOT wait or poll; continue " + "Runs in the background: dispatch returns immediately with live transcript paths, and each completed unit " + "re-enters the conversation on its own — an ungrouped task as soon as IT finishes, tasks sharing a `group` " + "together once all of them finish. Handle each result as it lands. Do NOT wait or poll; continue " "other work. While children run, `action` (list/steer/stop) controls them live — steer when a transcript shows a " "child drifting.\n\n" "USE FOR: reasoning-heavy subtasks, work that would flood your context with intermediate data, or independent " @@ -546,6 +547,13 @@ DELEGATE_TASK_SCHEMA = { "schema_valid, plus schema_errors on failure). Keep it forgiving — require only " "fields you will read.", ), + "group": _p( + "string", + "Optional completion group. Tasks sharing a group wait for each other and return as ONE " + "message (use when you must compare or merge their results); a task without a group " + "returns on its own the moment it finishes. Independent work (separate PR reviews, " + "unrelated fixes) should stay ungrouped so nothing waits for the slowest sibling.", + ), }, "required": ["goal"], }, diff --git a/tools/delegate_tool_dispatch.py b/tools/delegate_tool_dispatch.py index 03093da0e6..6d21dc566a 100644 --- a/tools/delegate_tool_dispatch.py +++ b/tools/delegate_tool_dispatch.py @@ -1,6 +1,7 @@ """Batch execution + background dispatch for delegate_task: ``delegate_task`` builds a ``_Batch`` -(children + origin identity) and hands it to ``_run_batch``, which runs it synchronously or as ONE -detached async unit.""" +(children + origin identity) and hands it to ``_run_batch``, which runs it synchronously or as +detached async units — one per task ``group`` (ungrouped tasks are a unit each), so a finished +review does not wait for the slowest sibling.""" from __future__ import annotations @@ -9,7 +10,7 @@ import json import logging import time from concurrent.futures import FIRST_COMPLETED, wait as _cf_wait -from dataclasses import dataclass +from dataclasses import dataclass, replace from typing import Any, Dict, List, Optional from tools.delegate_tool_child_run import _detach_child, _fabricated_entry, _signal_child_stop @@ -42,6 +43,8 @@ class _Batch: origin_owner_transport: Any origin_owner_session_record: Any overall_start: float + # Set on per-group units carved out by ``_dispatch_background``; None for the whole batch / ungrouped units. + group: Optional[str] = None def owner_kwargs(self) -> Dict[str, Any]: """Steer/stop authority of the originating session, passed to every child run.""" @@ -103,6 +106,7 @@ def _run_children_parallel(batch: _Batch, results: list, *, honor_parent_interru _tag = format_batch_tag(batch.live_deleg_id, parent_agent) # Fabricated entries for still-pending / raised futures carry the correct _delegate_role. _child_by_index = {i: child for (i, _, child) in batch.children} + n_here = len(batch.children) # a per-group unit runs a subset; ``n_tasks`` keeps the call-wide ``i/N`` slot def _entry_of(future, idx): if not future.done(): @@ -125,17 +129,17 @@ def _run_children_parallel(batch: _Batch, results: list, *, honor_parent_interru for future in done: entry = _entry_of(future, futures[future]) results.append(entry) - _report_child_done(parent_agent, spinner_ref, entry, _tag, task_labels, n_tasks, n_tasks - len(results)) + _report_child_done(parent_agent, spinner_ref, entry, _tag, task_labels, n_tasks, n_here - len(results)) results.sort(key=lambda r: r["task_index"]) # match input order def _execute_and_aggregate(batch: _Batch, *, honor_parent_interrupt: bool = True) -> dict: - """Run all built children, join, finalize (hooks + cost rollup), return the combined dict. Shared by the sync path - and the background runner: even in the background the batch JOINS on itself here so ONE consolidated results - block re-enters the conversation. Live transcripts are finalized but retained as the full-fidelity record - (retention pruning happens on future dispatches).""" + """Run the batch's built children, join, finalize (hooks + cost rollup), return the combined dict. Shared by the + sync path and the background runner; a background runner receives a per-group unit (a subset of the call's + children) so each group JOINS only on itself. Live transcripts are finalized but retained as the full-fidelity + record (retention pruning happens on future dispatches).""" from tools.delegation_live_log import update_manifest_statuses results: list = [] - if len(batch.task_list) == 1: + if len(batch.children) == 1: results.append(batch.run_child(*batch.children[0])) else: _run_children_parallel(batch, results, honor_parent_interrupt=honor_parent_interrupt) @@ -152,8 +156,11 @@ def _execute_and_aggregate(batch: _Batch, *, honor_parent_interrupt: bool = True update_manifest_statuses(batch.live_deleg_id, results) combined: Dict[str, Any] = {"results": results, "total_duration_seconds": total_duration} - if batch.live_paths: - combined["live_transcripts"] = list(batch.live_paths) + unit_paths = [batch.live_paths[i] for (i, _, _) in batch.children if i < len(batch.live_paths)] + if unit_paths: + combined["live_transcripts"] = unit_paths + if batch.group is not None: + combined["group"] = batch.group return combined _SYNC_FALLBACK_NOTES = { @@ -256,9 +263,10 @@ _BACKGROUND_NOTES = { "conversation as a new message when it finishes. Do not wait or poll — just continue." ), "many": ( - "{n} subagents are running in parallel in the background. You and the user can keep working; they wait on " - "each other and their consolidated results re-enter the conversation as a single message once ALL of them " - "finish. Do not wait or poll — just continue." + "{n} subagents are running in parallel in the background as {k} independent unit(s). You and the user can keep " + "working; each unit's results re-enter the conversation as their own message as soon as THAT unit finishes " + "(tasks sharing a `group` finish together; ungrouped tasks report individually), so act on each as it lands. " + "Do not wait or poll — just continue." ), "control_hint": ( "While a child runs you can orchestrate it live with this same tool: delegate_task(action='list') to see live " @@ -271,29 +279,66 @@ _BACKGROUND_NOTES = { ), } -def _dispatched_payload(dispatch: dict, goals: List[str], child_agents: List[Any], live_paths: List[str]) -> dict: - """Model-facing handle for an accepted background batch.""" +def _dispatched_payload(batch: _Batch, units: List[tuple[_Batch, str]]) -> dict: + """Model-facing handle for an accepted background call: one entry per async unit.""" + goals = [t["goal"] for t in batch.task_list] n = len(goals) payload = { - "status": "dispatched", "mode": "background", "count": n, "delegation_id": dispatch["delegation_id"], - "goals": goals, "note": _BACKGROUND_NOTES["one"] if n == 1 else _BACKGROUND_NOTES["many"].format(n=n), + "status": "dispatched", "mode": "background", "count": n, + "delegation_id": batch.live_deleg_id or units[0][1], "goals": goals, + "note": _BACKGROUND_NOTES["one"] if n == 1 else _BACKGROUND_NOTES["many"].format(n=n, k=len(units)), } - sids = [getattr(c, "_subagent_id", None) for c in child_agents] + if len(units) > 1: + payload["units"] = [ + {"delegation_id": uid, "group": unit.group, "task_indexes": [i for (i, _, _) in unit.children]} + for unit, uid in units + ] + sids = [getattr(c, "_subagent_id", None) for (_, _, c) in batch.children] if any(isinstance(s, str) and s for s in sids): payload["subagent_ids"] = sids payload["control_hint"] = _BACKGROUND_NOTES["control_hint"] - if live_paths: - payload["live_transcripts"] = list(live_paths) + if batch.live_paths: + payload["live_transcripts"] = list(batch.live_paths) payload["live_transcripts_hint"] = _BACKGROUND_NOTES["live_transcripts_hint"] return payload -def _dispatch_background(batch: _Batch) -> str: - """Dispatch the WHOLE batch as one async unit and return the tool result JSON. The runner joins on every child and - yields ONE consolidated results block that re-enters the conversation as a single message when ALL children - finish. Falls back to running synchronously (with an explanatory ``note``) when the session cannot receive - detached completions or the async pool is at capacity.""" - from tools.delegate_tool import _get_max_async_children +def _units_of(batch: _Batch) -> List[_Batch]: + """Partition the call's children into async units: one per distinct task ``group`` (first-appearance order) and + one per ungrouped task. Each unit is a ``_Batch`` sharing the call's task_list/transcripts but owning a subset of + ``children``, so a unit joins only on itself and its completion re-enters the conversation on its own.""" + members: Dict[Any, List[tuple]] = {} + for i, t, c in batch.children: + g = t.get("group") + key = ("g", str(g)) if g not in (None, "") else ("i", i) + members.setdefault(key, []).append((i, t, c)) + return [replace(batch, children=ch, group=(key[1] if key[0] == "g" else None)) for key, ch in members.items()] + +def _dispatch_unit(unit: _Batch, unit_id: Optional[str], slot_key: Optional[str], routing: dict) -> dict: + """Hand ONE unit to the async registry; the runner joins on that unit's children only.""" from tools.async_delegation import dispatch_async_delegation_batch + child_agents = [c for (_, _, c) in unit.children] + + def _interrupt(): + for c in child_agents: + _signal_child_stop(c, "Async delegation cancelled") + + return dispatch_async_delegation_batch( + # Call-wide goals: completion formatting indexes them by task_index. + goals=[t["goal"] for t in unit.task_list], context=unit.context, + toolsets=None, # metadata for the completion block only; subagents inherit the parent's toolsets + role=unit.top_role, model=unit.creds["model"], + runner=lambda: _execute_and_aggregate(unit, honor_parent_interrupt=False), + interrupt_fn=_interrupt, delegation_id=unit_id, slot_key=slot_key, + task_indexes=[i for (i, _, _) in unit.children] if len(unit.children) < len(unit.task_list) else None, + progress_fn=lambda: _batch_progress_token(child_agents), **routing, + ) + +def _dispatch_background(batch: _Batch) -> str: + """Dispatch the call as independent async units (see ``_units_of``) and return the tool result JSON. Every unit + of one call shares ONE pool slot (``slot_key``), so grouping never changes capacity accounting. Falls back to + running synchronously (with an explanatory ``note``) when the session cannot receive detached completions or the + async pool is at capacity.""" + from tools.delegate_tool import _get_max_async_children wake_sid = _resolve_async_wake_sid(batch.origin_wake_sid) if wake_sid is None: logger.info("delegate_task: async delivery unsupported on this session runtime; running the batch synchronously instead.") @@ -301,40 +346,42 @@ def _dispatch_background(batch: _Batch) -> str: parent_agent = batch.parent_agent session_key, origin_ui_session_id = _resolve_async_session_key(parent_agent, batch.origin_ui_session_id) - child_agents = [c for (_, _, c) in batch.children] - # The batch's lifecycle is owned by the async registry now: drop the children from the parent's + # The children's lifecycle is owned by the async registry now: drop them from the parent's # interrupt-propagation list (_build_child_agent attached them, which is correct for sync runs). - for c in child_agents: + for (_, _, c) in batch.children: _detach_child(parent_agent, c) - - def _batch_interrupt(): - # Cancellation path for the detached batch (owned by the async registry). - for c in child_agents: - _signal_child_stop(c, "Async delegation cancelled") - - goals = [t["goal"] for t in batch.task_list] - dispatch = dispatch_async_delegation_batch( - goals=goals, context=batch.context, - toolsets=None, # metadata for the completion block only; subagents inherit the parent's toolsets - role=batch.top_role, model=batch.creds["model"], session_key=session_key, - origin_ui_session_id=origin_ui_session_id, origin_session_id=wake_sid, - parent_session_id=getattr(parent_agent, "session_id", None), - runner=lambda: _execute_and_aggregate(batch, honor_parent_interrupt=False), - interrupt_fn=_batch_interrupt, max_async_children=_get_max_async_children(), - # Reuse the live-transcript directory's id (when created) so the returned delegation_id matches - # cache/delegation/live//. - delegation_id=batch.live_deleg_id, - progress_fn=lambda: _batch_progress_token(child_agents), + routing = dict( + session_key=session_key, origin_ui_session_id=origin_ui_session_id, origin_session_id=wake_sid, + parent_session_id=getattr(parent_agent, "session_id", None), max_async_children=_get_max_async_children(), ) - if dispatch.get("status") == "dispatched": - return json.dumps(_dispatched_payload(dispatch, goals, child_agents, batch.live_paths), ensure_ascii=False) - # Pool at capacity / schedule failure: the async unit was never accepted, so just run inline (re-attaching to the - # parent list is not needed). - logger.info( - "delegate_task: async pool at capacity (%s); running the whole batch synchronously instead.", - dispatch.get("error", "rejected"), - ) - return _run_sync_with_note(batch, "at_capacity") + + units = _units_of(batch) + dispatched: List[tuple[_Batch, str]] = [] + inline_results: List[dict] = [] + slot_key: Optional[str] = None + for k, unit in enumerate(units): + # One unit keeps the live-transcript directory's id so the returned delegation_id matches + # cache/delegation/live//; several units suffix it (-1, -2, ...) and the call keeps the bare id. + unit_id = batch.live_deleg_id if len(units) == 1 else (f"{batch.live_deleg_id}-{k + 1}" if batch.live_deleg_id else None) + dispatch = _dispatch_unit(unit, unit_id, slot_key, routing) + if dispatch.get("status") == "dispatched": + slot_key = slot_key or dispatch["delegation_id"] + dispatched.append((unit, dispatch["delegation_id"])) + continue + if not dispatched: + logger.info( + "delegate_task: async pool at capacity (%s); running the whole batch synchronously instead.", + dispatch.get("error", "rejected"), + ) + return _run_sync_with_note(batch, "at_capacity") + # Later units of an admitted call share its slot and cannot be capacity-rejected; a scheduler failure runs + # the unit inline so no task is silently dropped. + logger.warning("delegate_task: unit %d/%d not accepted (%s); running it inline.", k + 1, len(units), dispatch.get("error")) + inline_results.extend(_execute_and_aggregate(unit, honor_parent_interrupt=False)["results"]) + payload = _dispatched_payload(batch, dispatched) + if inline_results: + payload["inline_results"] = inline_results + return json.dumps(payload, ensure_ascii=False) def _run_batch(batch: _Batch, background: bool) -> str: """Tool result JSON: a dispatch handle (background) or the joined combined results.""" diff --git a/tools/process_registry_notifications.py b/tools/process_registry_notifications.py index 15c9355867..12944f0160 100644 --- a/tools/process_registry_notifications.py +++ b/tools/process_registry_notifications.py @@ -116,14 +116,16 @@ def _preamble(evt: dict, title: str, intro: str, completed_at: float, *, with_go def _format_batch_delegation(evt: dict, deleg_id: str, completed_at: float) -> str: """Consolidated block for a delegate_task fan-out that finished as one unit.""" results, goals = evt.get("results") or [], evt.get("goals") or [] - n = len(results) if results else len(goals) + # ``goals`` is the whole delegate_task call (task_index indexes it); ``results`` is this unit's subset. + n, n_unit = len(goals) or len(results), len(results) or len(goals) + group = evt.get("group") + unit = f"group '{group}' ({n_unit} subagent(s))" if group is not None else f"{n_unit} subagent(s)" lines = _preamble( evt, f"[ASYNC DELEGATION BATCH COMPLETE — {deleg_id}]", - f"A background fan-out of {n} subagent(s) you dispatched earlier " - "has finished. All ran in parallel and waited on each other; their " - "consolidated results are below. You may have moved on since " - "dispatching — act on these or re-dispatch if things have changed.", + f"A background fan-out unit you dispatched earlier — {unit} — has finished; its consolidated results are " + "below. Other units from the same delegate_task call (other groups / ungrouped tasks) report separately as " + "they finish. You may have moved on since dispatching — act on these or re-dispatch if things have changed.", completed_at, with_goal=False) lines[-1] += f" Total duration: {evt.get('total_duration_seconds', evt.get('duration_seconds', '?'))}s" if evt.get("error") and not results: diff --git a/website/docs/user-guide/features/delegation.md b/website/docs/user-guide/features/delegation.md index a4ce8aba3b..e417e55fbe 100644 --- a/website/docs/user-guide/features/delegation.md +++ b/website/docs/user-guide/features/delegation.md @@ -117,12 +117,26 @@ delegate_task( ## Batch Mode Details -When a top-level agent provides a `tasks` array, Hermes returns one background handle, runs the subagents in parallel, and posts one consolidated result after every child finishes. An orchestrator subagent waits for its batch in the current turn so it can synthesize the results. +When a top-level agent provides a `tasks` array, Hermes returns one background handle and runs the subagents in parallel. Results come back **per completion unit**, not once at the end: + +- A task **without** a `group` is its own unit: its result re-enters the conversation the moment that subagent finishes, so five independent PR reviews land as five messages and the agent acts on each without waiting for the slowest one. +- Tasks that share a `group` string wait for each other and return as **one** consolidated message (use this when the parent must compare or merge their outputs). + +```json +{"tasks": [ + {"goal": "Review PR #101 ..."}, + {"goal": "Review PR #102 ..."}, + {"goal": "Benchmark approach A ...", "group": "bench"}, + {"goal": "Benchmark approach B ...", "group": "bench"} +]} +``` + +The dispatch handle lists each unit (`units[].delegation_id`, `group`, `task_indexes`); unit ids are the call's id suffixed `-1`, `-2`, …, and every unit of one call shares a single slot of `delegation.max_concurrent_children`, so grouping never changes capacity accounting. An orchestrator subagent waits for its whole batch in the current turn so it can synthesize the results. - **Maximum concurrency:** 3 tasks by default (configurable via `delegation.max_concurrent_children` or the `DELEGATION_MAX_CONCURRENT_CHILDREN` env var; floor of 1, no hard ceiling). Batches larger than the limit return a tool error rather than being silently truncated. - **Thread pool:** Uses `ThreadPoolExecutor` with the configured concurrency limit as max workers - **Progress display:** In CLI mode, a tree-view shows tool calls from each subagent in real-time with per-task completion lines. In gateway mode, progress is batched and relayed to the parent's progress callback -- **Result ordering:** Results are sorted by task index to match input order regardless of completion order +- **Result ordering:** Within a unit, results are sorted by task index to match input order regardless of completion order; `TASK i/N` labels index the whole call - **Cancellation:** Follow-up messages do not cancel a top-level background batch. `/stop` or closing/resetting the owning session cancels its active children. Synchronous orchestrator children still follow their parent's interrupt state Synchronous single-task delegation from an orchestrator runs directly without thread pool overhead. From e9c527eeb656a14b8f038d8b57bc41e9fa6962ee Mon Sep 17 00:00:00 2001 From: Jerry Gooch Date: Sun, 6 Sep 2026 02:54:55 -0700 Subject: [PATCH 135/276] fix(desktop): cascade a sash drag through sibling panes past their floors MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A sash drag used to stop the moment its seam partner hit its min size, so a row of panes behaved like unrelated boxes: growing the Browser tile could not take space from Chat once the pane between them was at its floor. The drag now plans the whole run from the pointerdown sizes — the partner donates first, then its next visible sibling, and so on — and commits once on release (fixed zones get px overrides, the flex run gets weights). A reverse drag after a cascade stays local to the seam. Release only folds a tool zone that THIS gesture took to its floor, so an unrelated rail already resting there (a minimized Terminal) no longer cancels the Files/Chat commit. Salvaged from #103166 (Jerry Gooch), trimmed to the sash-drag change; the width/height lock feature, zone-body context menu and the restored-tool CSS floor are held as separate decisions. --- .../tree-split-cascade-resize.test.tsx | 183 ++++++++++++ .../pane-shell/tree/renderer/tree-split.tsx | 270 ++++++++++++------ 2 files changed, 367 insertions(+), 86 deletions(-) create mode 100644 apps/desktop/src/components/pane-shell/tree/renderer/tree-split-cascade-resize.test.tsx diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split-cascade-resize.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split-cascade-resize.test.tsx new file mode 100644 index 0000000000..b5a77ce5fa --- /dev/null +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split-cascade-resize.test.tsx @@ -0,0 +1,183 @@ +import { cleanup, fireEvent, render } from '@testing-library/react' +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' + +import { registry } from '@/contrib/registry' +import { $paneStates } from '@/store/panes' + +import { group, split, type SplitNode } from '../model' +import { $hiddenTreePanes, $layoutTree, markCollapsePane, setTreeGroupMinimized } from '../store' + +import { TreeSplit } from './tree-split' + +class TestResizeObserver { + observe() {} + unobserve() {} + disconnect() {} +} + +const disposers: (() => void)[] = [] + +beforeAll(() => { + vi.stubGlobal('ResizeObserver', TestResizeObserver) + vi.stubGlobal('CSS', { ...globalThis.CSS, escape: (value: string) => value }) + vi.stubGlobal('requestAnimationFrame', () => 1) + vi.stubGlobal('cancelAnimationFrame', () => undefined) + Element.prototype.hasPointerCapture ??= () => false + Element.prototype.setPointerCapture ??= () => undefined + Element.prototype.releasePointerCapture ??= () => undefined +}) + +beforeEach(() => { + window.localStorage.clear() + $hiddenTreePanes.set(new Set()) + $paneStates.set({}) + + disposers.push( + registry.register({ area: 'panes', data: { placement: 'main' }, id: 'chat', render: () => null, title: 'Chat' }), + registry.register({ + area: 'panes', + data: { placement: 'main', width: '100px' }, + id: 'cron', + render: () => null, + title: 'Cron' + }), + registry.register({ + area: 'panes', + data: { placement: 'main' }, + id: 'browser', + render: () => null, + title: 'Browser' + }) + ) +}) + +afterEach(() => { + cleanup() + $layoutTree.set(null) + $paneStates.set({}) + disposers.splice(0).forEach(dispose => dispose()) +}) + +function rect(width: number): DOMRect { + return { + bottom: 600, + height: 600, + left: 0, + right: width, + toJSON: () => ({}), + top: 0, + width, + x: 0, + y: 0 + } as DOMRect +} + +function setWidth(element: HTMLElement, width: number) { + Object.defineProperty(element, 'getBoundingClientRect', { configurable: true, value: () => rect(width) }) +} + +function row(): SplitNode { + const tree = $layoutTree.get() + + if (!tree || tree.type !== 'split') { + throw new Error('expected root row split') + } + + return tree +} + +describe('TreeSplit cascading expansion', () => { + it('grows Browser through Cron into Chat after Cron reaches its minimum', () => { + const tree = split( + 'row', + [ + group(['chat'], { id: 'chat-zone' }), + group(['cron'], { id: 'cron-zone' }), + group(['browser'], { id: 'browser-zone' }) + ], + [5, 1, 2], + 'root-row' + ) + + $layoutTree.set(tree) + + render() + + const container = document.querySelector('[data-tree-split="root-row"]')! + const [chat, cron, browser] = [...container.children] as HTMLElement[] + setWidth(container, 800) + setWidth(chat, 500) + setWidth(cron, 100) + setWidth(browser, 200) + setWidth(document.querySelector('[data-tree-group="cron-zone"]')!, 100) + + const browserSash = document.querySelectorAll('[role="separator"]')[1]! + fireEvent.pointerDown(browserSash, { button: 0, clientX: 600, pointerId: 1, pointerType: 'mouse' }) + fireEvent.pointerMove(window, { clientX: 300, pointerId: 1, pointerType: 'mouse' }) + fireEvent.pointerUp(window, { clientX: 300, pointerId: 1, pointerType: 'mouse' }) + + // Browser's 300px requested growth first takes Cron from 100px to its + // 80px floor, then takes the remaining 280px from Chat. The browser gets + // every released pixel instead of stopping at Cron's local floor. + expect($paneStates.get().cron?.widthOverride).toBe(80) + expect(row().weights).toEqual([2.2, 1, 5]) + }) + it('commits a regular cascade when an unrelated tool rail is already minimized', () => { + markCollapsePane('terminal') + disposers.push( + registry.register({ + area: 'panes', + data: { maxWidth: '600px', minWidth: '160px', placement: 'right', width: '200px' }, + id: 'browser', + render: () => null, + title: 'Browser' + }), + registry.register({ + area: 'panes', + data: { placement: 'bottom' }, + id: 'terminal', + render: () => null, + title: 'Terminal' + }) + ) + + const tree = split( + 'row', + [ + group(['chat'], { id: 'chat-zone' }), + group(['cron'], { id: 'cron-zone' }), + group(['browser'], { id: 'browser-zone' }), + group(['terminal'], { id: 'terminal-zone' }) + ], + [5, 1, 2, 0.28], + 'root-row' + ) + + $layoutTree.set(tree) + $paneStates.set({ browser: { open: true, widthOverride: 200 } }) + setTreeGroupMinimized('terminal-zone', true) + + render() + + const container = document.querySelector('[data-tree-split="root-row"]')! + const [chat, cron, browser, terminal] = [...container.children] as HTMLElement[] + setWidth(container, 828) + setWidth(chat, 500) + setWidth(cron, 100) + setWidth(browser, 200) + setWidth(terminal, 28) + setWidth(document.querySelector('[data-tree-group="cron-zone"]')!, 100) + setWidth(document.querySelector('[data-tree-group="browser-zone"]')!, 200) + setWidth(document.querySelector('[data-tree-group="terminal-zone"]')!, 28) + + const browserSash = document.querySelectorAll('[role="separator"]')[1]! + fireEvent.pointerDown(browserSash, { button: 0, clientX: 600, pointerId: 1, pointerType: 'mouse' }) + fireEvent.pointerMove(window, { clientX: 300, pointerId: 1, pointerType: 'mouse' }) + fireEvent.pointerUp(window, { clientX: 300, pointerId: 1, pointerType: 'mouse' }) + + expect($paneStates.get().cron?.widthOverride).toBe(80) + expect($paneStates.get().browser?.widthOverride).toBe(500) + expect(row().weights[0]).toBeCloseTo(2.2) + expect(row().children[3]).toMatchObject({ id: 'terminal-zone', minimized: true }) + }) +}) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx index 15b44edf02..f147b855ea 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx @@ -73,6 +73,7 @@ function useSubtreeOverrides(paneIds: readonly string[]): TrackContext['override const snapshot = useCallback(() => { const all = $paneStates.get() + const sig = paneIds.map(id => `${id}:${all[id]?.widthOverride ?? ''}:${all[id]?.heightOverride ?? ''}`).join('|') if (cache.current.sig !== sig) { @@ -245,15 +246,167 @@ export function TreeSplit({ node, root, rootRow }: { node: SplitNode; root?: boo return } - const a = sideFor(node.children[aIndex], kidA, 'end') - const b = sideFor(node.children[bIndex], kidB, 'start') - const a0px = a.fixed ? a.size : sizeOf(kidA) - const b0px = b.fixed ? b.size : sizeOf(kidB) - const lo = Math.max(a.min - a0px, b0px - b.max) - const hi = Math.min(a.max - a0px, b0px - b.min) + const tracks = node.children.map((child, index) => { + const element = container.children[index] as HTMLElement | undefined + if (!element) { + return null + } + + const side = sideFor(child, element, index <= aIndex ? 'end' : 'start') + + return { + ...side, + element, + index, + initial: side.fixed ? side.size : sizeOf(element), + // A minimized rail is its 28px strip: it neither donates nor takes, + // and its remembered weight must survive the gesture so restoring + // it brings back the size it had before it was folded. + minimized: child.type === 'group' && Boolean(child.minimized), + visible: !isCollapsed(child) + } + }) + + if (tracks.some(track => !track)) { + return + } + + const sashTracks = tracks as Array> const setOverride = horizontal ? setPaneWidthOverride : setPaneHeightOverride + // The seam is ONE boundary of a whole run: the side the pointer moves + // toward gives up space, its seam partner first. When the partner hits + // its floor the drag keeps going and takes from the partner's next + // visible sibling, and so on — a row behaves like one workspace instead + // of a set of unrelated boxes. Plans are absolute (from the sizes at + // pointerdown), so every frame is a pure function of the pointer. + const planFor = (requestedShift: number, allowCascade: boolean) => { + const toward = requestedShift >= 0 ? 1 : -1 + const targetIndex = toward > 0 ? aIndex : bIndex + // The partner is the nearest VISIBLE sibling (`aIndex`/`bIndex`), not + // `target ± 1` — a hidden zone parked between them (the closed preview + // pane) is display:none and must neither donate nor block. + const partnerIndex = toward > 0 ? bIndex : aIndex + const target = sashTracks[targetIndex] + const next = sashTracks.map(track => track.initial) + let remaining = Math.min(Math.abs(requestedShift), Math.max(0, target.max - target.initial)) + let transferred = 0 + let cascaded = false + // A tool rail is locally resizable at its own seam, but must not make + // every unrelated seam in the row local-only. It also remains the + // terminal donor when reached only after another regular track. + const cascade = allowCascade && !target.collapseId + + for (let donorIndex = partnerIndex; donorIndex >= 0 && donorIndex < sashTracks.length; donorIndex += toward) { + const donor = sashTracks[donorIndex] + + if (!donor.visible) { + continue + } + + if (donor.collapseId && donorIndex !== partnerIndex) { + break + } + + const take = Math.min(remaining, Math.max(0, donor.initial - donor.min)) + next[donorIndex] -= take + transferred += take + remaining -= take + + if (take > 0 && donorIndex !== partnerIndex) { + cascaded = true + } + + if (remaining === 0 || !cascade) { + break + } + } + + next[targetIndex] += transferred + + return { cascaded, moved: Math.sign(requestedShift) * transferred, sizes: next } + } + + // One store commit on release: fixed zones get their px override + // (sidebar semantics), the flex run gets weights. Clamped px → weights + // so persisted weights match what's on screen. + const commitPlan = (plan: ReturnType) => { + const weights = [...node.weights] + let weightsChanged = false + + sashTracks.forEach((track, index) => { + if (!track.visible || track.minimized) { + return + } + + const px = Math.round(plan.sizes[index]) + + if (track.fixed) { + if (px !== Math.round(track.initial)) { + track.paneIds.forEach(id => setOverride(id, px)) + } + } else { + weights[index] = Math.max(0.01, plan.sizes[index] / pxPerWeight) + weightsChanged = true + } + }) + + if (weightsChanged) { + setTreeSplitWeights(node.id, weights) + } + } + + const styleSnapshots = sashTracks.map(track => track.element.getAttribute('style')) + + const restoreStyles = () => { + sashTracks.forEach((track, index) => { + const style = styleSnapshots[index] + + if (style === null) { + track.element.removeAttribute('style') + } else { + track.element.setAttribute('style', style) + } + }) + } + + // During the gesture the store is NOT written. setTreeSplitWeights / + // setPaneWidthOverride each mint a new tree/pane-state object, and the + // resulting commit walks every mounted pane — measured live on a real + // 2-session layout: 31 commits across a 58-frame drag, 20.7fps, with + // TreeNode at 490ms and Block/Ct re-parsing markdown for 620ms. The + // store is written ONCE on release; during the drag the run is + // previewed with inline styles on the same wrappers React sizes. + // + // Preview rules (learned the hard way — a wrong shape here left a + // phantom gap where a hidden sidebar lived): + // - a FIXED track gets ONLY a flex-basis override. Its wrapper renders + // as `flex: 0 1 `, so basis is the whole difference; grow and + // shrink stay React's. + // - every visible flex track is pinned to `0 1 `. The plan keeps + // the run's total px constant, so no leftover gap can open. + // - hidden (display:none) and minimized tracks are left untouched. + // - cleanup: a real drag commits the store once, and React's re-render + // rewrites the `flex` shorthand, which clears the overrides (writing + // the shorthand resets the longhands). A no-movement click restores + // the captured style attributes instead, since nothing re-renders. + const previewPlan = (plan: ReturnType) => { + sashTracks.forEach((track, index) => { + if (!track.visible || track.minimized) { + return + } + + const px = plan.sizes[index] + + if (track.fixed) { + track.element.style.flexBasis = `${px}px` + } else { + track.element.style.flex = `0 1 ${px}px` + } + }) + } + try { handle.setPointerCapture?.(pointerId) } catch { @@ -272,70 +425,26 @@ export function TreeSplit({ node, root, rootRow }: { node: SplitNode; root?: boo // pointermove outpaces 60fps and each write relayouts the whole pane tree, // so coalesce to one apply per frame (rafCoalesce commits on cleanup). - // - // During the gesture the store is NOT written. setTreeSplitWeights / - // setPaneWidthOverride each mint a new tree/pane-state object, and the - // resulting commit walks every mounted pane — measured live on a real - // 2-session layout: 31 commits across a 58-frame drag, 20.7fps, with - // TreeNode at 490ms and Block/Ct re-parsing markdown for 620ms. The - // store is written ONCE on release; during the drag the seam is - // previewed with inline styles on the same wrappers React sizes. - // - // Preview rules (learned the hard way — a wrong shape here left a - // phantom gap where a hidden sidebar lived): - // - a FIXED side gets ONLY a flex-basis override. Its wrapper renders - // as `flex: 0 1 `, so basis is the whole difference; grow and - // shrink stay React's. Crucially the flex partner is left untouched, - // so it keeps absorbing the remainder and no leftover gap can open. - // - a flex-vs-flex seam pins both sides to `0 1 `. Their combined - // px is constant, so sibling flex tracks see the same leftover. - // - cleanup: a real drag commits the store once, and React's re-render - // rewrites the `flex` shorthand, which clears the overrides (writing - // the shorthand resets the longhands). A no-movement click restores - // the captured style attribute instead, since nothing re-renders. - const applyShift = (shiftPx: number) => { - if (a.fixed) { - a.paneIds.forEach(id => setOverride(id, Math.round(a0px + shiftPx))) - } - - if (b.fixed) { - b.paneIds.forEach(id => setOverride(id, Math.round(b0px - shiftPx))) - } - - if (!a.fixed && !b.fixed) { - const weights = [...node.weights] - // Clamped px → weights so persisted weights match what's on screen. - weights[aIndex] = (a0px + shiftPx) / pxPerWeight - weights[bIndex] = (b0px - shiftPx) / pxPerWeight - setTreeSplitWeights(node.id, weights) - } - } - - const styleA = kidA.getAttribute('style') - const styleB = kidB.getAttribute('style') - - const previewSide = (el: HTMLElement, fixed: boolean, px: number) => { - if (fixed) { - el.style.flexBasis = `${px}px` - } else if (!a.fixed && !b.fixed) { - el.style.flex = `0 1 ${px}px` - } - // Mixed seam, flex side: untouched — it absorbs what the fixed side - // gives up, exactly as the track model would render it. - } - - const previewShift = (shiftPx: number) => { - previewSide(kidA, a.fixed, a0px + shiftPx) - previewSide(kidB, b.fixed, b0px - shiftPx) - } - - const resize = rafCoalesce(previewShift) - let lastShift: null | number = null + const resize = rafCoalesce(previewPlan) + // A return drag stays local only after this gesture has actually crossed + // a second donor. A one-pixel opposite wobble must not poison the real + // forward movement that follows. + let cascadeDirection: -1 | 1 | null = null + let lastPlan: null | ReturnType = null let done = false const onMove = (ev: PointerEvent) => { - lastShift = Math.max(lo, Math.min(hi, (horizontal ? ev.clientX : ev.clientY) - start)) - resize.push(lastShift) + const shift = (horizontal ? ev.clientX : ev.clientY) - start + const direction = Math.sign(shift) as -1 | 0 | 1 + + const allowCascade = cascadeDirection === null || direction === cascadeDirection + lastPlan = planFor(shift, allowCascade) + + if (cascadeDirection === null && direction !== 0 && lastPlan.cascaded) { + cascadeDirection = direction + } + + resize.push(lastPlan) } // Ends through several racing paths (pointerup, pointercancel, window @@ -349,36 +458,25 @@ export function TreeSplit({ node, root, rootRow }: { node: SplitNode; root?: boo done = true resize.finish() - if (lastShift !== null) { + if (lastPlan && lastPlan.moved !== 0) { // Dragged a tool panel down to its collapsed header? Fold the zone // to its rail instead of persisting a sliver — and DON'T write the // sliver size, so restoring brings back the size it had before. - const collapsedSide = - (a.collapseId && a0px + lastShift <= a.floor && a.collapseId) || - (b.collapseId && b0px - lastShift <= b.floor && b.collapseId) || - null + // Only a track THIS gesture took to its floor counts: an unrelated + // rail already resting there must not cancel the commit. + const collapsedSide = sashTracks.find( + (track, index) => track.collapseId && track.initial > track.floor && lastPlan!.sizes[index] <= track.floor + )?.collapseId if (collapsedSide) { setTreeGroupMinimized(collapsedSide, true) } else { - // One store commit; the re-render rewrites `flex` and clears the - // preview overrides. - applyShift(lastShift) + commitPlan(lastPlan) } } else { // Click without movement: nothing will re-render, so put the // wrappers' inline styles back exactly as React last wrote them. - if (styleA === null) { - kidA.removeAttribute('style') - } else { - kidA.setAttribute('style', styleA) - } - - if (styleB === null) { - kidB.removeAttribute('style') - } else { - kidB.setAttribute('style', styleB) - } + restoreStyles() } // Geometry vars re-enable AFTER the final store commit above, so the From 817c0ce08acb8ebdefe29deb6f01cd6efd12837b Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:54:55 -0700 Subject: [PATCH 136/276] fix(desktop): restore drag previews before the release commit folds a zone When a sash drag folds a tool zone to its rail, the store commit re-renders only the wrappers whose style prop changed; the flex sibling the gesture had pinned to `flex: 0 1 ` kept that inline preview and could not grow into the freed space. Restore every captured style attribute before the commit, on every release path. --- .../tree-split-cascade-resize.test.tsx | 50 +++++++++++++++++++ .../pane-shell/tree/renderer/tree-split.tsx | 18 ++++--- 2 files changed, 60 insertions(+), 8 deletions(-) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split-cascade-resize.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split-cascade-resize.test.tsx index b5a77ce5fa..22d256c8b7 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split-cascade-resize.test.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split-cascade-resize.test.tsx @@ -76,6 +76,13 @@ function setWidth(element: HTMLElement, width: number) { Object.defineProperty(element, 'getBoundingClientRect', { configurable: true, value: () => rect(width) }) } +function setHeight(element: HTMLElement, height: number) { + Object.defineProperty(element, 'getBoundingClientRect', { + configurable: true, + value: () => ({ ...rect(1000), bottom: height, height }) + }) +} + function row(): SplitNode { const tree = $layoutTree.get() @@ -180,4 +187,47 @@ describe('TreeSplit cascading expansion', () => { expect(row().weights[0]).toBeCloseTo(2.2) expect(row().children[3]).toMatchObject({ id: 'terminal-zone', minimized: true }) }) + it('folding a tool zone at its floor leaves no drag preview pinned on the flex sibling', () => { + markCollapsePane('terminal') + disposers.push( + registry.register({ + area: 'panes', + data: { height: '200px', placement: 'bottom' }, + id: 'terminal', + render: () => null, + title: 'Terminal' + }) + ) + + const tree = split( + 'column', + [group(['chat'], { id: 'chat-zone' }), group(['terminal'], { id: 'terminal-zone' })], + [1, 1], + 'root-column' + ) + + $layoutTree.set(tree) + + render() + + const container = document.querySelector('[data-tree-split="root-column"]')! + const [chat, terminal] = [...container.children] as HTMLElement[] + setHeight(container, 800) + setHeight(chat, 600) + setHeight(terminal, 200) + setHeight(document.querySelector('[data-tree-group="terminal-zone"]')!, 200) + + const chatFlex = chat.style.flex + const terminalSash = document.querySelectorAll('[role="separator"]')[0]! + fireEvent.pointerDown(terminalSash, { button: 0, clientY: 600, pointerId: 1, pointerType: 'mouse' }) + fireEvent.pointerMove(window, { clientY: 790, pointerId: 1, pointerType: 'mouse' }) + fireEvent.pointerUp(window, { clientY: 790, pointerId: 1, pointerType: 'mouse' }) + + // The zone folds to its rail (no sliver persisted) and the chat wrapper — + // which the commit does not re-render — is back on React's own flex, not + // the `0 1 ` pin the gesture previewed. + expect(row().children[1]).toMatchObject({ id: 'terminal-zone', minimized: true }) + expect($paneStates.get().terminal?.heightOverride).toBeUndefined() + expect(chat.style.flex).toBe(chatFlex) + }) }) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx index f147b855ea..f51069d6ed 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-split.tsx @@ -387,10 +387,8 @@ export function TreeSplit({ node, root, rootRow }: { node: SplitNode; root?: boo // - every visible flex track is pinned to `0 1 `. The plan keeps // the run's total px constant, so no leftover gap can open. // - hidden (display:none) and minimized tracks are left untouched. - // - cleanup: a real drag commits the store once, and React's re-render - // rewrites the `flex` shorthand, which clears the overrides (writing - // the shorthand resets the longhands). A no-movement click restores - // the captured style attributes instead, since nothing re-renders. + // - cleanup: release restores the captured style attributes, then a + // real drag commits the store once and React re-renders from it. const previewPlan = (plan: ReturnType) => { sashTracks.forEach((track, index) => { if (!track.visible || track.minimized) { @@ -458,6 +456,14 @@ export function TreeSplit({ node, root, rootRow }: { node: SplitNode; root?: boo done = true resize.finish() + // Put every wrapper's inline style back exactly as React last wrote + // it BEFORE the store commit. React only rewrites a wrapper whose + // style prop changed; a preview pinned on a track the commit leaves + // alone (the flex run beside a zone that folded to its rail) would + // otherwise survive as a stale `flex: 0 1 ` and stop it growing. + // A no-movement click has no commit, so this is also its whole cleanup. + restoreStyles() + if (lastPlan && lastPlan.moved !== 0) { // Dragged a tool panel down to its collapsed header? Fold the zone // to its rail instead of persisting a sliver — and DON'T write the @@ -473,10 +479,6 @@ export function TreeSplit({ node, root, rootRow }: { node: SplitNode; root?: boo } else { commitPlan(lastPlan) } - } else { - // Click without movement: nothing will re-render, so put the - // wrappers' inline styles back exactly as React last wrote them. - restoreStyles() } // Geometry vars re-enable AFTER the final store commit above, so the From e3a634a13c5c3e68c352cf7b7592770ba0ad7952 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:55:45 -0700 Subject: [PATCH 137/276] chore(contributors): map jerry@jerrygooch.com to jerrygooch --- contributors/emails/jerry@jerrygooch.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/jerry@jerrygooch.com diff --git a/contributors/emails/jerry@jerrygooch.com b/contributors/emails/jerry@jerrygooch.com new file mode 100644 index 0000000000..8fe05c85dc --- /dev/null +++ b/contributors/emails/jerry@jerrygooch.com @@ -0,0 +1,2 @@ +jerrygooch +# PR #103166 salvage From cacf1218d7dcd2c111f8b3282ec6e0df7fac0a02 Mon Sep 17 00:00:00 2001 From: "Axl Ibiza, MBA" Date: Sat, 5 Sep 2026 10:54:22 -0500 Subject: [PATCH 138/276] fix(desktop-update): acknowledge Windows progress completion A delayed browser could miss the 900ms terminal event and spin forever after the updater exited. Retain terminal delivery until the page acknowledges it, bound unavailable-client teardown and failed requests, and preserve a truthful final display. Fixes #103747. Builds on OutThisLife and Teknium detached handoff work in #83634 and the #75895 quiet-window design. Continues Axl Ibiza Windows update investigation (#60233, #94107, #100763), including source/review contributions carried by merged #93353 and #85170. Existing #102373, #103140, #95719, #97299 and #103632 retain their separate scopes. --- .../scripts/desktop-update-ui.test.mjs | 100 ++++++++++++++++++ scripts/desktop-update/ui.html | 46 ++++++-- scripts/desktop-update/windows.ps1 | 40 ++++++- ...test_desktop_update_windows_ui_delivery.py | 92 ++++++++++++++++ 4 files changed, 266 insertions(+), 12 deletions(-) create mode 100644 apps/desktop/scripts/desktop-update-ui.test.mjs create mode 100644 tests/test_desktop_update_windows_ui_delivery.py diff --git a/apps/desktop/scripts/desktop-update-ui.test.mjs b/apps/desktop/scripts/desktop-update-ui.test.mjs new file mode 100644 index 0000000000..bc69320a4a --- /dev/null +++ b/apps/desktop/scripts/desktop-update-ui.test.mjs @@ -0,0 +1,100 @@ +import assert from 'node:assert/strict' +import fs from 'node:fs' +import { JSDOM } from 'jsdom' +import { afterEach, test, vi } from 'vitest' + +// Execute the shipped page, including its inline script, rather than matching +// source strings or testing a second implementation of the progress client. +const html = fs.readFileSync(new URL('../../../scripts/desktop-update/ui.html', import.meta.url), 'utf8') +const windows = [] + +function openPage(fetch) { + vi.useFakeTimers() + const dom = new JSDOM(html, { + url: 'http://127.0.0.1:12345/', + runScripts: 'dangerously', + beforeParse(window) { + window.fetch = fetch + window.AbortController = AbortController + window.setTimeout = setTimeout + window.clearTimeout = clearTimeout + window.requestAnimationFrame = () => 1 + window.cancelAnimationFrame = () => {} + } + }) + windows.push(dom.window) + return dom.window.document +} + +afterEach(() => { + windows.splice(0).forEach(window => window.close()) + vi.useRealTimers() +}) + +test.each(['done', 'manual', 'error'])('renders %s before acknowledging terminal delivery', async status => { + let document + const requests = [] + const receipt = '550e8400-e29b-41d4-a716-446655440000' + const fetch = vi.fn(async (url, options) => { + requests.push(url) + if (url.startsWith('/ack/')) { + assert.equal(options.method, 'POST') + assert.equal(document.body.className, status === 'error' ? 'error' : 'done') + assert.notEqual(document.getElementById('title').textContent, 'Updating Hermes') + return { ok: true } + } + return { ok: true, json: async () => ({ status, receipt, message: 'The updater result' }) } + }) + document = openPage(fetch) + await vi.advanceTimersByTimeAsync(1000) + assert.equal(document.body.className, status === 'error' ? 'error' : 'done') + assert.notEqual(document.getElementById('title').textContent, 'Updating Hermes') + assert.deepEqual(requests, ['/progress', `/ack/${receipt}`]) +}) + +test.each(['disconnect', 'hung', 'hung-body', 'http', 'invalid'])('bounds %s progress failures without inventing an update outcome', async failure => { + let attempts = 0 + const fetch = vi.fn((_url, options) => { + attempts++ + if (attempts === 1) { + return Promise.resolve({ ok: true, json: async () => ({ status: 'running', message: 'Installing dependencies' }) }) + } + if (failure === 'hung' || failure === 'hung-body') { + const pending = () => new Promise((_resolve, reject) => { + options.signal?.addEventListener('abort', () => reject(new Error('timeout')), { once: true }) + }) + return failure === 'hung' ? pending() : Promise.resolve({ ok: true, json: pending }) + } + if (failure === 'http') return Promise.resolve({ ok: false }) + if (failure === 'invalid') return Promise.resolve({ ok: true, json: async () => ({}) }) + return Promise.reject(new Error('connection refused')) + }) + const document = openPage(fetch) + await vi.advanceTimersByTimeAsync(20_000) + assert.equal(document.body.className, 'disconnected') + assert.equal(document.getElementById('title').textContent, 'Update status unavailable') + assert.match(document.getElementById('line').textContent, /Check Hermes/) + assert.ok(attempts <= 4, `unbounded retry loop: ${attempts}`) +}) + +test.each(['legacy', 'transient', 'ack-failure'])('preserves terminal truth with %s servers', async mode => { + let attempts = 0 + const fetch = vi.fn(async url => { + if (url.startsWith('/ack/')) throw new Error('server already stopped') + if (++attempts === 1 && mode === 'transient') throw new Error('temporary disconnect') + return { ok: true, json: async () => ({ status: 'done', ...(mode === 'ack-failure' ? { receipt: 'test-receipt' } : {}) }) } + }) + const document = openPage(fetch) + await vi.advanceTimersByTimeAsync(20_000) + assert.equal(document.body.className, 'done') + assert.equal(document.getElementById('title').textContent, 'Update complete') +}) + +test('continues displaying a healthy long update while progress remains reachable', async () => { + const fetch = vi.fn(async () => ({ ok: true, json: async () => ({ status: 'running', message: 'Building Desktop' }) })) + const document = openPage(fetch) + await vi.advanceTimersByTimeAsync(60_000) + assert.equal(document.body.className, '') + assert.equal(document.getElementById('title').textContent, 'Updating Hermes') + assert.equal(document.getElementById('line').textContent, 'Building Desktop') +}) diff --git a/scripts/desktop-update/ui.html b/scripts/desktop-update/ui.html index 6e55cb694c..3b0646b260 100644 --- a/scripts/desktop-update/ui.html +++ b/scripts/desktop-update/ui.html @@ -99,8 +99,8 @@ font-size: 11px; color: var(--foreground); } - body.done #loader, body.error #loader { display: none; } - body.done #glyph, body.error #glyph { display: flex; } + body.done #loader, body.error #loader, body.disconnected #loader { display: none; } + body.done #glyph, body.error #glyph, body.disconnected #glyph { display: flex; } @@ -205,6 +205,7 @@ const glyphEl = document.getElementById('glyph') const defaultLine = lineEl.textContent /* what a stage-less run says */ let settled = false + let failures = 0 const elapsedText = s => s < 60 ? `${s}s elapsed` : `${Math.floor(s / 60)}m ${s % 60}s elapsed` @@ -226,7 +227,8 @@ } else if (state.status === 'done') { settle('done') glyphEl.textContent = '\u2713' - lineEl.textContent = 'Opening Hermes\u2026' + titleEl.textContent = 'Update complete' + lineEl.textContent = 'Opening Hermes\u2026\nYou can close this window.' } else if (state.status === 'manual') { // Update landed but Hermes will NOT reopen itself (package skew, // sandbox helper, launch rejected). The orchestrator leaves this @@ -243,13 +245,43 @@ } } + async function request(url, options = {}) { + const controller = new AbortController() + const timeout = setTimeout(() => controller.abort(), 5000) + try { + const response = await fetch(url, { ...options, cache: 'no-store', signal: controller.signal }) + if (!response.ok) throw new Error('Progress request failed') + // Keep the deadline active while reading the body too: receiving + // headers alone does not prove that the progress server is responsive. + return options.method === 'POST' ? null : await response.json() + } finally { + clearTimeout(timeout) + } + } + async function poll() { try { - const res = await fetch('/progress', { cache: 'no-store' }) - if (res.ok) apply(await res.json()) + const state = await request('/progress') + if (!state || !['running', 'done', 'manual', 'error'].includes(state.status)) { + throw new Error('Invalid progress response') + } + failures = 0 + apply(state) + // The Windows server can now wait for delivery instead of guessing + // that one polling interval was enough. Older/POSIX servers omit this + // receipt. Apply first so a failed acknowledgement cannot hide a result. + if (settled && typeof state.receipt === 'string' && state.receipt) { + await request(`/ack/${encodeURIComponent(state.receipt)}`, { method: 'POST' }) + } } catch { - // Server gone: hold the last known state. The orchestrator owns - // closing this window; the relaunched Desktop owns the result. + if (!settled && ++failures >= 3) { + // A vanished server is not evidence of success OR failure. Leave an + // honest, finite state even if the browser refuses the close request. + settle('disconnected') + glyphEl.textContent = '!' + titleEl.textContent = 'Update status unavailable' + lineEl.textContent = 'The progress connection was lost.\nCheck Hermes for the update result. You can close this window.' + } } if (!settled) setTimeout(poll, 400) } diff --git a/scripts/desktop-update/windows.ps1 b/scripts/desktop-update/windows.ps1 index 19c9d8615a..320db0d845 100644 --- a/scripts/desktop-update/windows.ps1 +++ b/scripts/desktop-update/windows.ps1 @@ -108,6 +108,8 @@ $script:UiState = [hashtable]::Synchronized(@{ status = "running" # running | done | manual | error message = $script:UiStage clock = $script:UiStopwatch + receipt = $null + acknowledged_receipt = $null }) $script:UiServer = $null # @{ Listener; Runspace; PowerShell; Port; BrowserProc; Profile } @@ -190,15 +192,26 @@ function Start-UiServer([string]$HtmlPath) { $request = $reader.ReadLine() # Drain headers so the client doesn't see a reset mid-send. while ($true) { $h = $reader.ReadLine(); if ($null -eq $h -or $h -eq "") { break } } - if ($request -match "^GET /progress") { + if ($request -match "^GET /progress HTTP/1\.[01]$") { $elapsed = [Math]::Floor($State.clock.Elapsed.TotalSeconds) $snapshot = @{ status = $State.status message = $State.message elapsed_seconds = $elapsed + receipt = $State.receipt } | ConvertTo-Json -Compress Send-Response $stream "200 OK" "application/json; charset=utf-8" ([System.Text.Encoding]::UTF8.GetBytes($snapshot)) - } elseif ($request -match "^GET / ") { + } elseif ($request -match "^POST /ack/([^ /?]+) HTTP/1\.[01]$") { + $receipt = $Matches[1] + if ($State.status -in @("done", "manual", "error") -and $State.receipt -and $receipt -ceq $State.receipt) { + # Flush acceptance before waking the owner that will + # close the listener. No request body is needed. + Send-Response $stream "204 No Content" "text/plain" ([byte[]]@()) + $State.acknowledged_receipt = $receipt + } else { + Send-Response $stream "409 Conflict" "text/plain" ([System.Text.Encoding]::ASCII.GetBytes("unknown terminal receipt")) + } + } elseif ($request -match "^GET / HTTP/1\.[01]$") { Send-Response $stream "200 OK" "text/html; charset=utf-8" $HtmlBytes } else { Send-Response $stream "404 Not Found" "text/plain" ([System.Text.Encoding]::ASCII.GetBytes("not found")) @@ -283,11 +296,26 @@ function Stop-UiServer([switch]$LeaveWindow) { } function Publish-UiEvent([string]$Status, [string]$Message) { - # The event the shim listens for. One beat of poll latency (400ms) before - # teardown so the page actually renders the terminal state. + # A background browser can miss a fixed 900ms delivery window. Retain the + # terminal event until the page acknowledges applying this exact receipt. + # Older/headless clients cannot acknowledge, so teardown remains bounded. + $receipt = [Guid]::NewGuid().ToString('N') + $script:UiState.receipt = $receipt + $script:UiState.acknowledged_receipt = $null $script:UiState.message = $Message $script:UiState.status = $Status - if ($script:UiServer) { Start-Sleep -Milliseconds 900 } + if ($script:UiServer) { + $deliveryWait = [System.Diagnostics.Stopwatch]::StartNew() + while ($script:UiState.acknowledged_receipt -cne $receipt -and $deliveryWait.Elapsed.TotalSeconds -lt 10) { + Start-Sleep -Milliseconds 50 + if ($script:Ui) { [System.Windows.Forms.Application]::DoEvents() } + } + if ($script:UiState.acknowledged_receipt -ceq $receipt) { + Write-HandoffLog "shim: terminal state '$Status' acknowledged by the window" + } else { + Write-HandoffLog "shim: terminal state '$Status' was not acknowledged within 10s; closing the progress server" + } + } } function Get-UiElapsedText { @@ -309,6 +337,8 @@ function Publish-UiProgress([string]$Message) { $script:UiStage = $Message $script:UiState.message = $Message $script:UiState.status = "running" + $script:UiState.receipt = $null + $script:UiState.acknowledged_receipt = $null if ($script:Ui) { try { $script:Ui.Sub.Text = Get-UiProgressLine diff --git a/tests/test_desktop_update_windows_ui_delivery.py b/tests/test_desktop_update_windows_ui_delivery.py new file mode 100644 index 0000000000..dcd7b17c96 --- /dev/null +++ b/tests/test_desktop_update_windows_ui_delivery.py @@ -0,0 +1,92 @@ +"""The real Windows update server retains terminal events until acknowledged.""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import subprocess +import time +from contextlib import contextmanager +from pathlib import Path +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +import pytest + +pytestmark = pytest.mark.windows_only +SCRIPT = Path(__file__).resolve().parents[1] / "scripts/desktop-update/windows.ps1" + + +@contextmanager +def _server(tmp_path: Path, *, failed: bool = False): + powershell = shutil.which("powershell.exe") + assert powershell, "Windows updater tests require Windows PowerShell." + env = os.environ.copy() + env.update(TEMP=str(tmp_path), TMP=str(tmp_path), HERMES_SELFTEST_HOLD_SECONDS="0") + env.pop("HERMES_SELFTEST_FAIL", None) + if failed: + env["HERMES_SELFTEST_FAIL"] = "1" + output_path = tmp_path / "ui-delivery.log" + with output_path.open("wb") as output: + process = subprocess.Popen( + [powershell, "-NoProfile", "-ExecutionPolicy", "Bypass", "-File", str(SCRIPT), "-SelfTestUi", "-NoUi"], + env=env, + stdout=output, + stderr=subprocess.STDOUT, + creationflags=subprocess.CREATE_NO_WINDOW, + ) + try: + deadline = time.monotonic() + 30 + while time.monotonic() < deadline: + text = output_path.read_text(encoding="utf-8", errors="replace") + match = re.search(r"SELF-TEST: shim at (http://127\.0\.0\.1:\d+/)", text) + if match: + yield process, match.group(1) + return + if process.poll() is not None: + break + time.sleep(0.05) + pytest.fail(f"Update server did not publish a serving URL: {text}") + finally: + if process.poll() is None: + process.kill() + process.wait(timeout=5) + + +def _request(url: str, *, post: bool = False): + request = Request(url, data=b"" if post else None, method="POST" if post else "GET") + return urlopen(request, timeout=3) + + +@pytest.mark.parametrize("failed", [False, True], ids=["complete", "failed"]) +def test_delayed_client_receives_terminal_state_before_acknowledging(tmp_path: Path, failed: bool) -> None: + with _server(tmp_path, failed=failed) as (process, url): + # A background browser can miss the former 900ms terminal-state window. + time.sleep(2) + assert process.poll() is None, "the updater discarded its final state before the delayed client received it" + with _request(url + "progress") as response: + state = json.load(response) + assert state["status"] == ("error" if failed else "done") + receipt = state["receipt"] + assert isinstance(receipt, str) and receipt + + with pytest.raises(HTTPError) as wrong: + _request(url + "ack/not-the-published-receipt", post=True) + assert wrong.value.code == 409 + assert process.poll() is None, "an unrelated acknowledgement must not dispose the window's state" + + for route, post in [("progress-other", False), ("ack/" + receipt, False), ("ack/" + receipt + "/extra", True)]: + with pytest.raises(HTTPError) as invalid: + _request(url + route, post=post) + assert invalid.value.code == 404 + + with _request(url + "ack/" + receipt, post=True) as response: + assert response.status == 204 + assert process.wait(timeout=15) == 0 + + +def test_no_client_does_not_keep_the_update_process_alive_forever(tmp_path: Path) -> None: + with _server(tmp_path) as (process, _url): + assert process.wait(timeout=25) == 0 From a6a19fbe695041d8ae4212c7898d001b101e412b Mon Sep 17 00:00:00 2001 From: "Axl Ibiza, MBA" Date: Sat, 5 Sep 2026 11:07:05 -0500 Subject: [PATCH 139/276] test(ci): run desktop regressions when the shipped update page changes --- scripts/ci/classify_changes.py | 7 ++++++- tests/ci/test_classify_changes.py | 6 ++++++ 2 files changed, 12 insertions(+), 1 deletion(-) diff --git a/scripts/ci/classify_changes.py b/scripts/ci/classify_changes.py index 71afcda686..aee7d724a6 100644 --- a/scripts/ci/classify_changes.py +++ b/scripts/ci/classify_changes.py @@ -67,6 +67,8 @@ import subprocess import sys _FRONTEND = ("ui-tui/", "web/", "apps/") # TS typecheck-matrix packages +# Shipped page outside those packages, exercised by the desktop Electron suite. +_FRONTEND_FILES = {"scripts/desktop-update/ui.html"} _ROOT_NPM = {"package.json", "package-lock.json"} # shifts every package's tree _DOCKER_META = ("docker/", ".hadolint.yml", "Dockerfile") # docker setup _NIX_PATHS = ("nix/",) # nix files @@ -214,7 +216,10 @@ def classify(files: list[str]) -> dict[str, bool]: files = [f.strip() for f in files if f.strip()] python = any(not _py_irrelevant(f) for f in files) python_prod = any(not _py_irrelevant(f) and not _py_test_only(f) for f in files) - frontend = any(f.startswith(_FRONTEND) or f in _ROOT_NPM for f in files) + frontend = any( + f.startswith(_FRONTEND) or f in _ROOT_NPM or f in _FRONTEND_FILES + for f in files + ) deps = any(f == "pyproject.toml" for f in files) npm_lock = any(f.split("/")[-1] == "package-lock.json" for f in files) docker_meta = any(f.startswith(_DOCKER_META) for f in files) diff --git a/tests/ci/test_classify_changes.py b/tests/ci/test_classify_changes.py index 43e33dd620..a04a4dc4af 100644 --- a/tests/ci/test_classify_changes.py +++ b/tests/ci/test_classify_changes.py @@ -153,6 +153,12 @@ CASES = { ["scripts/desktop-update/windows.ps1"], _lanes(python=True, desktop_updater=True), ), + # The shipped updater page is exercised by the desktop Electron suite; + # a page-only change must run that suite as well as the server tests. + "updater ui.html → frontend + desktop_updater": ( + ["scripts/desktop-update/ui.html"], + _lanes(python=True, frontend=True, desktop_updater=True), + ), "desktop-update test → desktop_updater": ( ["tests/test_desktop_update_windows_progress.py"], _lanes(python=True, python_prod=False, scan=True, desktop_updater=True), From d024a9c4949559b8bd2fccd3798813976b48d089 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:40:48 -0700 Subject: [PATCH 140/276] fix(desktop-update): drop the dead WinForms pump from the ack wait Publish-UiEvent only enters the delivery-wait loop under $script:UiServer, and Show-ProgressWindow returns before the WinForms card is built whenever the browser shim is up, so $script:Ui is always null there. The DoEvents call could never run; the 50ms sleep loop is the whole wait. --- scripts/desktop-update/windows.ps1 | 1 - 1 file changed, 1 deletion(-) diff --git a/scripts/desktop-update/windows.ps1 b/scripts/desktop-update/windows.ps1 index 320db0d845..6c9bcc0167 100644 --- a/scripts/desktop-update/windows.ps1 +++ b/scripts/desktop-update/windows.ps1 @@ -308,7 +308,6 @@ function Publish-UiEvent([string]$Status, [string]$Message) { $deliveryWait = [System.Diagnostics.Stopwatch]::StartNew() while ($script:UiState.acknowledged_receipt -cne $receipt -and $deliveryWait.Elapsed.TotalSeconds -lt 10) { Start-Sleep -Milliseconds 50 - if ($script:Ui) { [System.Windows.Forms.Application]::DoEvents() } } if ($script:UiState.acknowledged_receipt -ceq $receipt) { Write-HandoffLog "shim: terminal state '$Status' acknowledged by the window" From 2f45f4d043add975845c47ef68fdf0be44573232 Mon Sep 17 00:00:00 2001 From: Halldrix <12357213+Halldrix@users.noreply.github.com> Date: Thu, 3 Sep 2026 05:55:21 -0500 Subject: [PATCH 141/276] fix(tui): drive /heartbeat firing from the session-owner process (#102056) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit /heartbeat is parsed by the slash-worker process, which starts its own heartbeat watchdog thread there — but that process never runs turns, so the queued prompt is armed-but-dead (countdown froze at ~0s, no turn ever injected). Mirror the existing /loop driver: poll for due heartbeats in the per-session notification poller and re-enter the live session through _run_prompt_submit as a plain user turn, preserving alternation and prompt caching. (cherry picked from commit 23ae971075f2d8156642f4c268dd010510a7af41) --- tests/tui_gateway/test_heartbeat_tui_tick.py | 106 +++++++++++++++++++ tui_gateway/session_notifications.py | 48 ++++++++- 2 files changed, 153 insertions(+), 1 deletion(-) create mode 100644 tests/tui_gateway/test_heartbeat_tui_tick.py diff --git a/tests/tui_gateway/test_heartbeat_tui_tick.py b/tests/tui_gateway/test_heartbeat_tui_tick.py new file mode 100644 index 0000000000..6b9d02f7cf --- /dev/null +++ b/tests/tui_gateway/test_heartbeat_tui_tick.py @@ -0,0 +1,106 @@ +"""Tests for the TUI/Desktop heartbeat driver (issue #102056). + +The slash worker that parses ``/heartbeat`` starts a watchdog thread in its +own process, where the queued prompt is never consumed. The real driver lives +in ``tui_gateway/server._maybe_fire_tui_heartbeat_tick``, which polls the +session-owner process and re-enters the live session via ``_run_prompt_submit``. +""" + +import time +from types import SimpleNamespace + +import pytest + +import tui_gateway.server as server +from hermes_cli.heartbeat import HeartbeatManager, HeartbeatState + + +def _due_manager(session_id): + """A HeartbeatManager whose state is already due, bypassing SessionDB.""" + mgr = HeartbeatManager.__new__(HeartbeatManager) + mgr.session_id = session_id + # created long ago so is_due() is immediately true. + mgr._state = HeartbeatState( + prompt="report backend health", + interval_seconds=60, + status="active", + created_at=time.time() - 3600, + ) + return mgr + + +def _idle_session(): + return { + "agent": SimpleNamespace(), + "session_key": "hb-tui-key", + "history": [], + "history_lock": server.threading.Lock(), + "running": False, + } + + +def test_tui_heartbeat_fires_when_idle_and_due(monkeypatch): + submitted = [] + + def _submit(_rid, sid, session, text): + submitted.append(text) + + monkeypatch.setattr(server, "_run_prompt_submit", _submit) + monkeypatch.setattr(server, "_emit", lambda *a, **k: None) + monkeypatch.setattr( + "hermes_cli.heartbeat.HeartbeatManager", lambda session_id: _due_manager(session_id) + ) + + session = _idle_session() + server._maybe_fire_tui_heartbeat_tick("sid-hb", session) + + assert len(submitted) == 1 + assert "report backend health" in submitted[0] + assert session["running"] is True + + +def test_tui_heartbeat_skips_when_busy(monkeypatch): + submitted = [] + + def _submit(_rid, sid, session, text): + submitted.append(text) + + monkeypatch.setattr(server, "_run_prompt_submit", _submit) + monkeypatch.setattr(server, "_emit", lambda *a, **k: None) + monkeypatch.setattr( + "hermes_cli.heartbeat.HeartbeatManager", lambda session_id: _due_manager(session_id) + ) + + session = _idle_session() + session["running"] = True # an in-flight turn owns the session + server._maybe_fire_tui_heartbeat_tick("sid-hb", session) + + assert submitted == [] + # The tick stays due; the driver never claims the session. + assert session["running"] is True + + +def test_tui_heartbeat_skips_when_not_due(monkeypatch): + submitted = [] + + def _submit(_rid, sid, session, text): + submitted.append(text) + + monkeypatch.setattr(server, "_run_prompt_submit", _submit) + monkeypatch.setattr(server, "_emit", lambda *a, **k: None) + + mgr = HeartbeatManager.__new__(HeartbeatManager) + mgr.session_id = "hb-tui-key" + mgr._state = HeartbeatState( + prompt="not yet", + interval_seconds=60, + status="active", + created_at=time.time(), # just armed — not due + ) + monkeypatch.setattr("hermes_cli.heartbeat.HeartbeatManager", lambda session_id: mgr) + + session = _idle_session() + server._maybe_fire_tui_heartbeat_tick("sid-hb", session) + + assert submitted == [] + assert session["running"] is False \ No newline at end of file diff --git a/tui_gateway/session_notifications.py b/tui_gateway/session_notifications.py index 3457bfa7ed..1cef2242be 100644 --- a/tui_gateway/session_notifications.py +++ b/tui_gateway/session_notifications.py @@ -113,6 +113,7 @@ def _notification_event_dedup_key(evt: dict) -> tuple: # past them and they can't wedge a later completed/blocked event behind an unclaimed row. _KANBAN_NOTIFY_KINDS = ("completed", "blocked", "gave_up", "crashed", "timed_out", "status", "archived", "unblocked") _KANBAN_POLL_SECONDS = _LOOP_POLL_SECONDS = 5.0 +_HEARTBEAT_POLL_SECONDS = 5.0 def _notif_release_turn(session: dict) -> None: @@ -173,6 +174,42 @@ def _notif_slash_loop_tick(rid: str, sid: str, session: dict, mgr, wakeup: str) _notif_loop_status(sid, decision["message"]) +def _maybe_fire_tui_heartbeat_tick(sid: str, session: dict) -> None: + """Fire a due /heartbeat prompt for an idle TUI/Desktop session (issue #102056). + + The slash-worker process that parses ``/heartbeat`` starts a heartbeat watchdog thread in its + OWN process, where the queued prompt has no turn loop to consume it — armed-but-dead. The + durable state lives in SessionDB, so the real driver belongs in the session-owner process: + this mirror of ``_maybe_fire_tui_loop_tick`` polls from the per-session notification poller + and re-enters the live session through ``_run_prompt_submit`` as a plain user turn (alternation + + prompt caching untouched). + """ + try: + from hermes_cli.heartbeat import HeartbeatManager + except Exception: + return + sid_key = session.get("session_key") or "" + if not sid_key: + return + mgr = HeartbeatManager(session_id=sid_key) + if not mgr.is_active() or not mgr.state.is_due(): + return + if not _notif_claim_turn(session): + return # busy — tick coalesces to the next idle poll + prompt = mgr.due_prompt() + if not prompt: + _notif_release_turn(session) + return + rid = f"__heartbeat__{int(time.time() * 1000)}" + try: + _emit("status.update", sid, {"kind": "heartbeat", "text": "♥ heartbeat firing…"}) + _emit("message.start", sid) + _run_prompt_submit(rid, sid, session, prompt) + except Exception as exc: + _notif_log_failure("heartbeat dispatch failed", exc) + _notif_release_turn(session) + + def _maybe_fire_tui_loop_tick(sid: str, session: dict) -> None: """Fire a due /loop wakeup for an idle TUI/Desktop/dashboard session (per-session poller, coarse cadence). Claims the session (running=True) before dispatching so a racing user prompt wins; the post-turn hook completes the tick.""" @@ -416,9 +453,18 @@ def _notification_poller_loop(stop_event: threading.Event, sid: str, session: di emitted: set = set() # dedup re-queued events so one completion isn't emitted 50 times while busy handle = lambda evt, deferred: _notif_handle_event( # noqa: E731 sid, session, evt, emitted, process_registry, format_process_notification, deferred) - last_kanban_poll = last_loop_poll = 0.0 + last_kanban_poll = last_loop_poll = last_heartbeat_poll = 0.0 while not stop_event.is_set() and not session.get("_finalized"): now = time.monotonic() + # ── /heartbeat driver ───────────────────────────────────────────────── + # The slash worker that parses /heartbeat cannot drive firing (its watchdog queues into a process that + # never runs turns), so the session-owner process polls for due heartbeats itself (#102056). + if now - last_heartbeat_poll >= _HEARTBEAT_POLL_SECONDS: + last_heartbeat_poll = now + try: + _maybe_fire_tui_heartbeat_tick(sid, session) + except Exception as hb_exc: + _notif_log_failure("heartbeat poll failed", hb_exc) # /loop wakeup driver: fire a due tick for THIS session while idle (same claim-under-lock as kanban dispatch). # An active non-parked /goal owns the idle boundary and defers it. if now - last_loop_poll >= _LOOP_POLL_SECONDS: From 5106e939e0b32d3cd70a6acf33943fd2d9ab58d7 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:32:51 -0700 Subject: [PATCH 142/276] fix(tui): heartbeat ticks share the /loop idle poll and rewind when no turn starts Folds the /heartbeat driver into the existing /loop poll slot (one cadence, one try/except table) and closes the consumed-tick gap: a dispatch that raises OR is refused by _admit_prompt_turn (returns False) releases the session claim and rewinds the persisted fire via HeartbeatManager.abandon_fire(), so the tick stays due for the next poll instead of advancing fire_count with no turn. abandon_fire refuses to overwrite a pause/resume/clear that landed between claim and dispatch (mirrors LoopManager.abandon_tick). Tests use the real SessionDB under a temp HERMES_HOME and drive the poller loop itself; docs note the TUI/Desktop surface. Rollback-on-failure idea credited to jerrygooch (#104011, 51df1ed4d04 / 283fa972169); the driver placement is Halldrix's (#102118). Reported in #102056 / #103044 (Vksh07). --- ...12357213+Halldrix@users.noreply.github.com | 2 + hermes_cli/heartbeat.py | 18 ++ tests/tui_gateway/test_heartbeat_tui_tick.py | 193 +++++++++++------- tui_gateway/session_notifications.py | 64 +++--- website/docs/user-guide/features/heartbeat.md | 2 +- 5 files changed, 164 insertions(+), 115 deletions(-) create mode 100644 contributors/emails/12357213+Halldrix@users.noreply.github.com diff --git a/contributors/emails/12357213+Halldrix@users.noreply.github.com b/contributors/emails/12357213+Halldrix@users.noreply.github.com new file mode 100644 index 0000000000..788e359ca4 --- /dev/null +++ b/contributors/emails/12357213+Halldrix@users.noreply.github.com @@ -0,0 +1,2 @@ +Halldrix +# PR #102118 /heartbeat TUI driver diff --git a/hermes_cli/heartbeat.py b/hermes_cli/heartbeat.py index ed5dddbace..d3f440d9ed 100644 --- a/hermes_cli/heartbeat.py +++ b/hermes_cli/heartbeat.py @@ -140,6 +140,7 @@ class HeartbeatManager: def __init__(self, session_id: str): self.session_id = session_id self._state: Optional[HeartbeatState] = load_heartbeat(session_id) + self._last_claim: Optional[tuple[float, int]] = None # (last_fired_at, fire_count) before the last due_prompt @property def state(self) -> Optional[HeartbeatState]: @@ -206,11 +207,28 @@ class HeartbeatManager: s = self._state if s is None or not s.is_due(now): return None + self._last_claim = (s.last_fired_at, s.fire_count) s.last_fired_at = now if now is not None else time.time() s.fire_count += 1 save_heartbeat(self.session_id, s) return s.render_prompt() + def abandon_fire(self) -> bool: + """Rewind the fire recorded by the last :meth:`due_prompt` whose turn never started, so the tick stays + due for the next poll instead of being silently consumed. Mirrors ``LoopManager.abandon_tick``. Skipped + (False) when the persisted state moved on — a pause/resume/clear that landed in between wins.""" + claim, s = self._last_claim, self._state + if claim is None or s is None: + return False + current = load_heartbeat(self.session_id) + if current is None or current.status != "active" or (current.last_fired_at, current.fire_count) != ( + s.last_fired_at, s.fire_count): + return False + s.last_fired_at, s.fire_count = claim + self._last_claim = None + save_heartbeat(self.session_id, s) + return True + def migrate_heartbeat_to_session(old_session_id: str, new_session_id: str) -> bool: """Carry a heartbeat across a compression session rotation (copy to child, archive parent, never raise). diff --git a/tests/tui_gateway/test_heartbeat_tui_tick.py b/tests/tui_gateway/test_heartbeat_tui_tick.py index 6b9d02f7cf..5052330d21 100644 --- a/tests/tui_gateway/test_heartbeat_tui_tick.py +++ b/tests/tui_gateway/test_heartbeat_tui_tick.py @@ -1,106 +1,145 @@ -"""Tests for the TUI/Desktop heartbeat driver (issue #102056). +"""/heartbeat firing from the TUI/Desktop session-owner process (#102056, #103044). -The slash worker that parses ``/heartbeat`` starts a watchdog thread in its -own process, where the queued prompt is never consumed. The real driver lives -in ``tui_gateway/server._maybe_fire_tui_heartbeat_tick``, which polls the -session-owner process and re-enters the live session via ``_run_prompt_submit``. +The slash worker that parses ``/heartbeat`` runs a HermesCLI whose watchdog queues the due prompt +into its own ``_pending_input`` — a queue no turn loop drains in that process. The per-session +notification poller (the same driver that fires ``/loop``) must poll the persisted HeartbeatManager +and re-enter the live session through ``_run_prompt_submit``. """ +from __future__ import annotations + +import importlib +import threading import time -from types import SimpleNamespace +from pathlib import Path +from unittest.mock import MagicMock, patch import pytest -import tui_gateway.server as server -from hermes_cli.heartbeat import HeartbeatManager, HeartbeatState + +@pytest.fixture() +def hermes_home(tmp_path, monkeypatch): + home = tmp_path / ".hermes" + home.mkdir() + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(home)) + from hermes_cli import goals + + goals._DB_CACHE.clear() + yield home + goals._DB_CACHE.clear() -def _due_manager(session_id): - """A HeartbeatManager whose state is already due, bypassing SessionDB.""" - mgr = HeartbeatManager.__new__(HeartbeatManager) - mgr.session_id = session_id - # created long ago so is_due() is immediately true. - mgr._state = HeartbeatState( - prompt="report backend health", - interval_seconds=60, - status="active", - created_at=time.time() - 3600, - ) +@pytest.fixture() +def server(hermes_home): + with patch.dict("sys.modules", {"hermes_cli.env_loader": MagicMock(), "hermes_cli.banner": MagicMock()}): + mod = importlib.import_module("tui_gateway.server") + yield mod + mod._sessions.clear() + + +@pytest.fixture() +def session(server): + sid, key = "sid-hb-test", "tui-hb-session-1" + s = {"session_key": key, "history": [], "history_lock": threading.Lock(), "history_version": 0, + "running": False, "attached_images": [], "cols": 120, "agent": MagicMock()} + server._sessions[sid] = s + return sid, key, s + + +def _arm_due(key: str): + from hermes_cli.heartbeat import HeartbeatManager, save_heartbeat + + mgr = HeartbeatManager(key) + state = mgr.set("report backend health", 60) + state.created_at = time.time() - 3600 + save_heartbeat(key, state) return mgr -def _idle_session(): - return { - "agent": SimpleNamespace(), - "session_key": "hb-tui-key", - "history": [], - "history_lock": server.threading.Lock(), - "running": False, - } +def _submits(server, submit): + return patch.object(server, "_run_prompt_submit", submit), patch.object(server, "_emit") -def test_tui_heartbeat_fires_when_idle_and_due(monkeypatch): - submitted = [] +def test_notification_poller_fires_due_heartbeat_when_idle(server, session): + """The session-owner poller loop itself dispatches a due heartbeat exactly once; the same state on + the base loop never fired (armed-but-dead).""" + sid, key, s = session + _arm_due(key) + dispatched: list[str] = [] - def _submit(_rid, sid, session, text): - submitted.append(text) + def submit(rid, sid_, session_, text, **kw): + dispatched.append(text) + return True - monkeypatch.setattr(server, "_run_prompt_submit", _submit) - monkeypatch.setattr(server, "_emit", lambda *a, **k: None) - monkeypatch.setattr( - "hermes_cli.heartbeat.HeartbeatManager", lambda session_id: _due_manager(session_id) - ) + stop = threading.Event() + p_submit, p_emit = _submits(server, submit) + with p_submit, p_emit: + t = threading.Thread(target=server._notification_poller_loop, args=(stop, sid, s), daemon=True) + t.start() + deadline = time.monotonic() + 8 + while not dispatched and time.monotonic() < deadline: + time.sleep(0.1) + stop.set() + t.join(timeout=5) - session = _idle_session() - server._maybe_fire_tui_heartbeat_tick("sid-hb", session) + from hermes_cli.heartbeat import load_heartbeat - assert len(submitted) == 1 - assert "report backend health" in submitted[0] - assert session["running"] is True + assert len(dispatched) == 1 and "report backend health" in dispatched[0] + assert s["running"] is True # claimed for the heartbeat turn + assert load_heartbeat(key).fire_count == 1 and not load_heartbeat(key).is_due() -def test_tui_heartbeat_skips_when_busy(monkeypatch): - submitted = [] +@pytest.mark.parametrize("running,due", [(True, True), (False, False)]) +def test_heartbeat_tick_defers_when_busy_or_not_due(server, session, running, due): + sid, key, s = session + mgr = _arm_due(key) + if not due: + mgr.state.created_at = time.time() + from hermes_cli.heartbeat import save_heartbeat - def _submit(_rid, sid, session, text): - submitted.append(text) + save_heartbeat(key, mgr.state) + s["running"] = running + p_submit, p_emit = _submits(server, MagicMock()) + with p_submit as submit, p_emit: + server._maybe_fire_tui_heartbeat_tick(sid, s) + submit.assert_not_called() + from hermes_cli.heartbeat import load_heartbeat - monkeypatch.setattr(server, "_run_prompt_submit", _submit) - monkeypatch.setattr(server, "_emit", lambda *a, **k: None) - monkeypatch.setattr( - "hermes_cli.heartbeat.HeartbeatManager", lambda session_id: _due_manager(session_id) - ) - - session = _idle_session() - session["running"] = True # an in-flight turn owns the session - server._maybe_fire_tui_heartbeat_tick("sid-hb", session) - - assert submitted == [] - # The tick stays due; the driver never claims the session. - assert session["running"] is True + assert s["running"] is running # a busy session is never released by the poller + assert load_heartbeat(key).fire_count == 0 # tick not consumed — still due when the session frees up -def test_tui_heartbeat_skips_when_not_due(monkeypatch): - submitted = [] +@pytest.mark.parametrize("failure", ["refused", "raised"]) +def test_heartbeat_dispatch_that_never_starts_a_turn_stays_due(server, session, failure): + """A refused/failed submit must release the claim AND rewind the persisted fire, otherwise the tick is + silently consumed (fire_count advanced, nothing ran) — the Desktop-side follow-up bug in #104011.""" + sid, key, s = session + _arm_due(key) - def _submit(_rid, sid, session, text): - submitted.append(text) + def submit(*a, **k): + if failure == "raised": + raise RuntimeError("transport down") + with s["history_lock"]: + s["running"] = False # _admit_prompt_turn refusal releases itself and returns False + return False - monkeypatch.setattr(server, "_run_prompt_submit", _submit) - monkeypatch.setattr(server, "_emit", lambda *a, **k: None) + p_submit, p_emit = _submits(server, submit) + with p_submit, p_emit: + server._maybe_fire_tui_heartbeat_tick(sid, s) - mgr = HeartbeatManager.__new__(HeartbeatManager) - mgr.session_id = "hb-tui-key" - mgr._state = HeartbeatState( - prompt="not yet", - interval_seconds=60, - status="active", - created_at=time.time(), # just armed — not due - ) - monkeypatch.setattr("hermes_cli.heartbeat.HeartbeatManager", lambda session_id: mgr) + from hermes_cli.heartbeat import load_heartbeat - session = _idle_session() - server._maybe_fire_tui_heartbeat_tick("sid-hb", session) + assert s["running"] is False + assert load_heartbeat(key).fire_count == 0 and load_heartbeat(key).is_due() - assert submitted == [] - assert session["running"] is False \ No newline at end of file + +def test_abandon_fire_never_overwrites_a_concurrent_pause(hermes_home): + from hermes_cli.heartbeat import HeartbeatManager, load_heartbeat + + key = "hb-abandon-race" + driver = _arm_due(key) + assert driver.due_prompt() is not None + HeartbeatManager(key).pause() # lands between the claim and the failed dispatch + assert driver.abandon_fire() is False + assert load_heartbeat(key).status == "paused" and load_heartbeat(key).fire_count == 1 diff --git a/tui_gateway/session_notifications.py b/tui_gateway/session_notifications.py index 1cef2242be..f00e1ec996 100644 --- a/tui_gateway/session_notifications.py +++ b/tui_gateway/session_notifications.py @@ -112,8 +112,7 @@ def _notification_event_dedup_key(evt: dict) -> tuple: # Mirror gateway/kanban_watchers.py TERMINAL_KINDS: claim silent kinds (archived/unblocked) too so the cursor advances # past them and they can't wedge a later completed/blocked event behind an unclaimed row. _KANBAN_NOTIFY_KINDS = ("completed", "blocked", "gave_up", "crashed", "timed_out", "status", "archived", "unblocked") -_KANBAN_POLL_SECONDS = _LOOP_POLL_SECONDS = 5.0 -_HEARTBEAT_POLL_SECONDS = 5.0 +_KANBAN_POLL_SECONDS = _LOOP_POLL_SECONDS = 5.0 # /loop and /heartbeat share one idle-poll cadence def _notif_release_turn(session: dict) -> None: @@ -175,39 +174,38 @@ def _notif_slash_loop_tick(rid: str, sid: str, session: dict, mgr, wakeup: str) def _maybe_fire_tui_heartbeat_tick(sid: str, session: dict) -> None: - """Fire a due /heartbeat prompt for an idle TUI/Desktop session (issue #102056). + """Fire a due /heartbeat prompt for an idle TUI/Desktop/dashboard session (#102056, #103044). - The slash-worker process that parses ``/heartbeat`` starts a heartbeat watchdog thread in its - OWN process, where the queued prompt has no turn loop to consume it — armed-but-dead. The - durable state lives in SessionDB, so the real driver belongs in the session-owner process: - this mirror of ``_maybe_fire_tui_loop_tick`` polls from the per-session notification poller - and re-enters the live session through ``_run_prompt_submit`` as a plain user turn (alternation - + prompt caching untouched). + ``/heartbeat`` runs in the slash worker, whose CLI watchdog queues the due prompt into a + ``_pending_input`` no turn loop ever drains — armed-but-dead. State is durable in SessionDB, so + the session-owner process drives firing exactly like ``_maybe_fire_tui_loop_tick``: claim the + idle session first (a racing user prompt wins), then re-enter through ``_run_prompt_submit`` as a + plain user turn. A dispatch that never starts a turn rewinds the persisted fire so the tick stays + due instead of being silently consumed. """ try: from hermes_cli.heartbeat import HeartbeatManager except Exception: return - sid_key = session.get("session_key") or "" - if not sid_key: + if not (sid_key := session.get("session_key") or ""): return mgr = HeartbeatManager(session_id=sid_key) - if not mgr.is_active() or not mgr.state.is_due(): - return - if not _notif_claim_turn(session): - return # busy — tick coalesces to the next idle poll - prompt = mgr.due_prompt() - if not prompt: + if not mgr.is_active() or not mgr.state.is_due() or not _notif_claim_turn(session): + return # not due, or busy — the tick coalesces to the next idle poll + if not (prompt := mgr.due_prompt()): _notif_release_turn(session) return - rid = f"__heartbeat__{int(time.time() * 1000)}" + started = False try: - _emit("status.update", sid, {"kind": "heartbeat", "text": "♥ heartbeat firing…"}) - _emit("message.start", sid) - _run_prompt_submit(rid, sid, session, prompt) + _emit("status.update", sid, {"kind": "heartbeat", "text": f"♥ heartbeat #{mgr.state.fire_count} firing…"}) + started = bool(_run_prompt_submit(f"__heartbeat__{int(time.time() * 1000)}", sid, session, prompt)) except Exception as exc: _notif_log_failure("heartbeat dispatch failed", exc) + if not started: + # _run_prompt_submit releases ``running`` itself when it refuses the turn; make it unconditional. _notif_release_turn(session) + with contextlib.suppress(Exception): + mgr.abandon_fire() def _maybe_fire_tui_loop_tick(sid: str, session: dict) -> None: @@ -453,26 +451,18 @@ def _notification_poller_loop(stop_event: threading.Event, sid: str, session: di emitted: set = set() # dedup re-queued events so one completion isn't emitted 50 times while busy handle = lambda evt, deferred: _notif_handle_event( # noqa: E731 sid, session, evt, emitted, process_registry, format_process_notification, deferred) - last_kanban_poll = last_loop_poll = last_heartbeat_poll = 0.0 + last_kanban_poll = last_loop_poll = 0.0 while not stop_event.is_set() and not session.get("_finalized"): now = time.monotonic() - # ── /heartbeat driver ───────────────────────────────────────────────── - # The slash worker that parses /heartbeat cannot drive firing (its watchdog queues into a process that - # never runs turns), so the session-owner process polls for due heartbeats itself (#102056). - if now - last_heartbeat_poll >= _HEARTBEAT_POLL_SECONDS: - last_heartbeat_poll = now - try: - _maybe_fire_tui_heartbeat_tick(sid, session) - except Exception as hb_exc: - _notif_log_failure("heartbeat poll failed", hb_exc) - # /loop wakeup driver: fire a due tick for THIS session while idle (same claim-under-lock as kanban dispatch). - # An active non-parked /goal owns the idle boundary and defers it. + # /loop and /heartbeat wakeup drivers: fire a due tick for THIS session while idle (same claim-under-lock + # as kanban dispatch). An active non-parked /goal owns the idle boundary and defers the loop tick. if now - last_loop_poll >= _LOOP_POLL_SECONDS: last_loop_poll = now - try: - _maybe_fire_tui_loop_tick(sid, session) - except Exception as loop_exc: - _notif_log_failure("loop wakeup poll failed", loop_exc) + for what, fire in (("loop wakeup", _maybe_fire_tui_loop_tick), ("heartbeat", _maybe_fire_tui_heartbeat_tick)): + try: + fire(sid, session) + except Exception as tick_exc: + _notif_log_failure(f"{what} poll failed", tick_exc) if now - last_kanban_poll >= _KANBAN_POLL_SECONDS: last_kanban_poll = now _notif_poll_kanban(sid, session) diff --git a/website/docs/user-guide/features/heartbeat.md b/website/docs/user-guide/features/heartbeat.md index 69984a3d7f..87c49e25f8 100644 --- a/website/docs/user-guide/features/heartbeat.md +++ b/website/docs/user-guide/features/heartbeat.md @@ -37,7 +37,7 @@ Rule of thumb: if the recurring prompt needs the conversation's context, use `/h | `/heartbeat resume` | Resume (re-anchors the timer — no instant stale fire). | | `/heartbeat clear` | Remove the heartbeat. | -`/hb` is an alias. Works on the CLI and gateway platforms (on Slack, use `/hermes heartbeat …`). +`/hb` is an alias. Works on the CLI, the TUI / Desktop app, and gateway platforms (on Slack, use `/hermes heartbeat …`). ## Behavior details From 77915e344cb0cd8e20661d4a7b393f987a2eef32 Mon Sep 17 00:00:00 2001 From: "Mark S." <274530371+unsupportedpastels@users.noreply.github.com> Date: Sun, 6 Sep 2026 11:54:01 +0000 Subject: [PATCH 143/276] fix(tui-gateway): preserve active isolated turns after disconnect --- .../test_isolated_orphan_activity.py | 207 ++++++++++++++++++ tui_gateway/compute_host.py | 24 +- tui_gateway/compute_host_bridge.py | 24 ++ tui_gateway/session_lifecycle.py | 7 + 4 files changed, 261 insertions(+), 1 deletion(-) create mode 100644 tests/tui_gateway/test_isolated_orphan_activity.py diff --git a/tests/tui_gateway/test_isolated_orphan_activity.py b/tests/tui_gateway/test_isolated_orphan_activity.py new file mode 100644 index 0000000000..fa130e011b --- /dev/null +++ b/tests/tui_gateway/test_isolated_orphan_activity.py @@ -0,0 +1,207 @@ +"""Detached Desktop/TUI turns use child-owned activity, not process heartbeats.""" + +from pathlib import Path +import sys +import threading +import time + +import pytest + +from tui_gateway import server +from tui_gateway.host_supervisor import HostSupervisor + + +class _Timer: + def __init__(self, delay, callback): + self.delay, self.callback = delay, callback + + def start(self): + pass + + def cancel(self): + pass + + +def _session(sid): + return dict(agent=None, agent_ready=threading.Event(), session_key=sid, + history=[], history_version=0, history_lock=threading.Lock(), + running=True, transport=server._detached_ws_transport, + attached_images=[], cols=80, source="desktop", inflight_turn=None) + + +@pytest.mark.parametrize("mode", ["fresh", "stale", "missing", "previous"]) +def test_real_child_detached_turn_activity(tmp_path, monkeypatch, mode): + """Real supervisor pipes, child admission/turn thread, bridge and orphan timer. + + Only the agent/provider and environment-heavy UI side effects are stubbed in + the child. Its activity writer and snapshot contract are the production ones. + """ + sid = "detached-turn" + session = _session(sid) + forwarded = [] + monkeypatch.setattr(server, "_sessions", {sid: session}) + monkeypatch.setattr(server, "_pending_ws_reaps", {}) + monkeypatch.setattr(server, "write_json", lambda msg: forwarded.append(msg) or True) + monkeypatch.setattr(server, "_load_dashboard_process_isolation_config", lambda: {"turn_isolation": True}) + monkeypatch.setattr(server, "_WS_ORPHAN_ACTIVITY_STALE_S", 30.0) + monkeypatch.setattr(server, "_WS_ORPHAN_REAP_GRACE_S", 20.0) + monkeypatch.setattr(server, "_session_has_active_delegations", lambda *args: False) + monkeypatch.setattr(server, "_session_cwd", lambda s: str(tmp_path)) + home = tmp_path / "home" + home.mkdir() + supervisor = HostSupervisor( + argv=[sys.executable, str(Path(__file__).resolve()), mode, str(tmp_path)], + registry_path=tmp_path / "host.json", env={"HERMES_HOME": str(home)}, + expected_hermes_home=str(home), rpc_sink=server._relay_compute_host_rpc, + heartbeat_secs=1, autostart=False) + monkeypatch.setattr(server, "_get_compute_host_supervisor", lambda *args: supervisor) + try: + response = server._submit_prompt_to_compute_host("request", sid, session, "work") + assert response["result"]["turn_isolation"] is True + deadline = time.monotonic() + 12 + while not (tmp_path / "provider-started").exists() and time.monotonic() < deadline: + time.sleep(0.02) + assert (tmp_path / "provider-started").exists(), supervisor._stderr_tail + # Give the actual child-to-parent sampler a bounded opportunity to arrive. + deadline = time.monotonic() + 3 + while not server._ws_orphan_turn_activity_is_fresh(session) and time.monotonic() < deadline: + time.sleep(0.02) + assert supervisor.is_running() + assert session["agent"] is None + assert server._ws_orphan_turn_activity_is_fresh(session) is (mode == "fresh") + monkeypatch.setattr(server.threading, "Timer", _Timer) + server._schedule_ws_orphan_reap(sid) + server._pending_ws_reaps[sid].callback() + assert bool(session.get("_client_gone_interrupt_requested")) is (mode != "fresh") + assert server._pending_ws_reaps[sid].delay == ( + 20.0 if mode == "fresh" else server._WS_ORPHAN_INTERRUPT_REAP_POLL_S) + assert not any(m.get("method") == "compute_host.activity" for m in forwarded) + if mode != "fresh": + deadline = time.monotonic() + 5 + while session["running"] and time.monotonic() < deadline: + time.sleep(0.02) + assert not session["running"], "stale child must receive and settle the real interrupt" + if mode == "fresh": + old_token = session["_compute_host_turn_id"] + old_request = next(iter(supervisor._pending_turns)) + (tmp_path / "release").touch() + deadline = time.monotonic() + 5 + while session["running"] and time.monotonic() < deadline: + time.sleep(0.02) + assert not session["running"] + assert "_compute_host_activity_ns" not in session + (tmp_path / "release").unlink() + (tmp_path / "provider-started").unlink() + session["running"] = True + # Same sid and caller rid, same child/agent, but NO new activity. + server._submit_prompt_to_compute_host("request", sid, session, "next") + assert session["_compute_host_turn_id"] != old_token + new_token = session["_compute_host_turn_id"] + # A delayed terminal frame cannot resolve the new caller-rid reuse. + supervisor._handle_host_frame({"type": "turn.end", "sid": sid, "request_id": old_request}) + assert session["running"] + assert session["_compute_host_turn_id"] == new_token + deadline = time.monotonic() + 5 + while not (tmp_path / "provider-started").exists() and time.monotonic() < deadline: + time.sleep(0.02) + assert (tmp_path / "provider-started").exists() + # Also replay a delayed sample from the previous dispatch. + server._relay_compute_host_rpc({"method": "compute_host.activity", "params": { + "session_id": sid, "turn_id": old_token, "activity_ns": time.perf_counter_ns()}}) + deadline = time.monotonic() + 3 + while "_compute_host_activity_ns" not in session and time.monotonic() < deadline: + time.sleep(0.02) + assert "_compute_host_activity_ns" in session + assert not server._ws_orphan_turn_activity_is_fresh(session) + server._pending_ws_reaps[sid].callback() + assert session["_client_gone_interrupt_requested"] + finally: + supervisor.shutdown() + + +@pytest.mark.parametrize("change", ["none", "other-session", "old-turn", "not-running", "stale", "missing"]) +def test_activity_relay_is_fenced_and_ages(monkeypatch, change): + session = _session("session") + session.update(_compute_host_active=True, _compute_host_turn_id="new-turn") + monkeypatch.setattr(server, "_sessions", {"session": session}) + monkeypatch.setattr(server, "_WS_ORPHAN_ACTIVITY_STALE_S", 30) + monkeypatch.setattr(server, "write_json", lambda msg: pytest.fail("internal activity leaked to client")) + params: dict = dict(session_id="session", turn_id="new-turn", activity_ns=time.perf_counter_ns()) + if change == "other-session": + params["session_id"] = "other" + elif change == "old-turn": + params["turn_id"] = "old-turn" + elif change == "not-running": + session["running"] = False + elif change == "stale": + params["activity_ns"] -= 31_000_000_000 + elif change == "missing": + params["activity_ns"] = None + server._relay_compute_host_rpc({"jsonrpc": "2.0", "method": "compute_host.activity", "params": params}) + assert server._ws_orphan_turn_activity_is_fresh(session) is (change == "none") + if change == "none": + # Repeated delivery is an observation of the same clock, not a refresh. + monkeypatch.setattr(server.time, "perf_counter_ns", lambda: params["activity_ns"] + 31_000_000_000) + server._relay_compute_host_rpc({"method": "compute_host.activity", "params": params}) + assert not server._ws_orphan_turn_activity_is_fresh(session) + + +def _run_child(mode, directory): + import socket + from agent.activity_tracking import ActivityTrackingMixin + from agent.session_activity import build_activity_snapshot + from tui_gateway.compute_host import run_host + + def no_network(*args, **kwargs): + raise AssertionError("test child must not contact a provider") + socket.socket.connect = no_network + + class Agent(ActivityTrackingMixin): + def __init__(self, sid): + self.session_id = sid + self._interrupt = threading.Event() + if mode == "previous": + self._touch_activity("previous turn") + + def get_activity_summary(self): + return build_activity_snapshot(last_activity_at=getattr(self, "_last_activity_ts", None), + last_activity_description="test provider") + + def clear_interrupt(self): + self._interrupt.clear() + + def interrupt(self, **kwargs): + self._interrupt.set() + + def run_conversation(self, *args, **kwargs): + Path(directory, "provider-started").touch() + deadline = time.monotonic() + 20 + while not self._interrupt.wait(0.05) and time.monotonic() < deadline: + if Path(directory, "release").exists(): + break + if mode == "fresh" and args[0] != "next": + self._touch_activity("provider wait") + elif mode == "stale": + self._last_activity_ts = time.time() - 3600 + return {"final_response": "done", "interrupted": self._interrupt.is_set()} + + def init(sid, key, agent, history, **kwargs): + s = _session(sid) + s.update(agent=agent, running=False, transport=None, + image_counter=0, slash_worker=None, show_reasoning=False, + tool_progress_mode="all") + server._sessions[sid] = s + + server._make_agent = lambda sid, *a, **kw: Agent(sid) + server._init_session = init + server._wire_callbacks = lambda *a: None + server._sync_agent_model_with_config = lambda *a: None + server._register_session_cwd = lambda *a: None + server._tts_stream_begin = lambda: None + server._sync_session_key_after_compress = lambda *a, **kw: None + server._get_usage = lambda *a: {} + run_host(stdout=sys.__stdout__) + + +if __name__ == "__main__": + _run_child(sys.argv[1], sys.argv[2]) diff --git a/tui_gateway/compute_host.py b/tui_gateway/compute_host.py index 48256fc709..2b76aa5b7b 100644 --- a/tui_gateway/compute_host.py +++ b/tui_gateway/compute_host.py @@ -7,6 +7,7 @@ import argparse import concurrent.futures import contextlib import json +import logging import os import signal import subprocess @@ -223,6 +224,7 @@ class ComputeHost: return session.update(running=True, _turn_cancel_requested=False, last_active=time.time()) server._start_inflight_turn(session, inflight) + turn_started_at = time.time() self._reply("turn.started", sid, request_id, started_ns=now_ns()) with contextlib.suppress(Exception): server._ensure_session_db_row(session) @@ -235,7 +237,10 @@ class ComputeHost: request_id, sid, session, text, display_kind=frame.get("display_kind") or None) run_thread = session.get("_run_thread") if run_thread is not None and hasattr(run_thread, "join"): - run_thread.join() + while run_thread.is_alive(): + run_thread.join(timeout=1.0) + if run_thread.is_alive() and frame.get("turn_id"): + self._emit_turn_activity(sid, session, frame["turn_id"], turn_started_at) with session["history_lock"]: meta = _history_meta(session) interrupted = bool(session.get("_turn_cancel_requested")) @@ -255,6 +260,23 @@ class ComputeHost: server._clear_inflight_turn(session) self._reply("turn.error", sid, request_id, reason="exception", message=str(exc)) + def _emit_turn_activity(self, sid: str, session: dict, turn_id: str, started_at: float) -> None: + # Observe the agent clock, never the host heartbeat. A reused agent's last + # turn must not lend its activity to a new turn that has not made progress. + activity_ns = None + try: + summary = session["agent"].get_activity_summary() + stamped_at = summary.get("last_activity_at") + elapsed = summary.get("seconds_since_activity") + if stamped_at is not None and stamped_at >= started_at and elapsed is not None and elapsed >= 0: + activity_ns = now_ns() - int(elapsed * 1_000_000_000) + except Exception: + logging.getLogger(__name__).debug("compute host activity unavailable sid=%s", sid, exc_info=True) + # perf_counter is shared across local processes; queued frames and cached + # samples age without requiring synchronized wall clocks in the parent. + self._transport.write({"jsonrpc": "2.0", "method": "compute_host.activity", "params": { + "session_id": sid, "turn_id": turn_id, "activity_ns": activity_ns}}) + def _ensure_server_session(self, server: Any, frame: dict[str, Any]) -> dict: sid = str(frame.get("sid") or "") session = server._sessions.get(sid) diff --git a/tui_gateway/compute_host_bridge.py b/tui_gateway/compute_host_bridge.py index 1804ce9da4..ab5fd9b022 100644 --- a/tui_gateway/compute_host_bridge.py +++ b/tui_gateway/compute_host_bridge.py @@ -91,6 +91,15 @@ def _compute_host_adopt_frame_meta(session: dict, frame: dict) -> None: def _relay_compute_host_rpc(message: dict) -> bool: """Relay host events while retaining the clarify snapshot needed on resume.""" params = message.get("params") if isinstance(message, dict) else None + if isinstance(message, dict) and message.get("method") == "compute_host.activity": + if isinstance(params, dict): + session = _sessions.get(str(params.get("session_id") or "")) + if session is not None: + with _history_lock(session): + if (session.get("running") and params.get("turn_id") + and session.get("_compute_host_turn_id") == params["turn_id"]): + session["_compute_host_activity_ns"] = params.get("activity_ns") + return True # Internal observation, not a client event or replay entry. kind = params.get("type") if isinstance(params, dict) else None if kind in {"clarify.request", "clarify.expire"}: session = _sessions.get(str(params.get("session_id") or "")) @@ -208,15 +217,30 @@ def _submit_prompt_to_compute_host( frame = _compute_host_turn_frame(rid, sid, session, text, image_paths=image_paths, queued_prompt_generation=queued_prompt_generation, display_kind=display_kind) + # Caller JSON-RPC ids may repeat across sockets and turns. Use an opaque + # dispatch lifetime token, installed before a fast child can send activity. + turn_id = frame["turn_id"] = frame["request_id"] = uuid.uuid4().hex + with session["history_lock"]: + session["_compute_host_turn_id"] = turn_id + session.pop("_compute_host_activity_ns", None) def _complete(done: dict) -> None: # submit_turn reports a synchronous pipe failure via the callback before re-raising; # leave the session untouched so prompt.submit can fail open to the in-process path. if done.get("reason") != "send_failed": + with session["history_lock"]: + if session.get("_compute_host_turn_id") != turn_id: + return + session.pop("_compute_host_turn_id", None) + session.pop("_compute_host_activity_ns", None) _on_compute_host_turn_done(rid, sid, session, done) try: _get_compute_host_supervisor(cfg).submit_turn(frame, on_complete=_complete) except Exception as exc: + with session["history_lock"]: + if session.get("_compute_host_turn_id") == turn_id: + session.pop("_compute_host_turn_id", None) + session.pop("_compute_host_activity_ns", None) return _err(rid, 5019, f"compute-host dispatch failed: {exc}") with session["history_lock"]: session["_compute_host_active"] = True diff --git a/tui_gateway/session_lifecycle.py b/tui_gateway/session_lifecycle.py index e7a77fcb37..0679f1ea28 100644 --- a/tui_gateway/session_lifecycle.py +++ b/tui_gateway/session_lifecycle.py @@ -473,9 +473,16 @@ def _ws_orphan_turn_activity_is_fresh(session: dict) -> bool: Reuses the agent's existing activity summary (``_touch_activity`` is stamped by API waits, stream tokens, and tool heartbeats — the same clock the turn-liveness watchdog samples; see agent/turn_liveness.py). See #100325, #98028. + Isolated turns mirror that clock from the child under a unique dispatch token; + their monotonic samples keep aging even if the child or its pipe stalls. """ if _WS_ORPHAN_ACTIVITY_STALE_S <= 0: return False + if session.get("_compute_host_turn_id"): + with session["history_lock"]: + stamp = session.get("_compute_host_activity_ns") + return (session.get("running", False) and isinstance(stamp, int) + and 0 <= (time.perf_counter_ns() - stamp) / 1_000_000_000 < _WS_ORPHAN_ACTIVITY_STALE_S) if not callable(summary_fn := getattr(session.get("agent"), "get_activity_summary", None)): return False try: From 8236f3768773a04b157d5f9acb112deaec733c7d Mon Sep 17 00:00:00 2001 From: RHODIZ IT Date: Thu, 3 Sep 2026 15:05:01 -0500 Subject: [PATCH 144/276] fix(cron): avoid reopening finalized session on teardown --- cron/scheduler.py | 9 ++++++ tests/cron/test_scheduler.py | 58 ++++++++++++++++++++++++++++++++++++ 2 files changed, 67 insertions(+) diff --git a/cron/scheduler.py b/cron/scheduler.py index 78b57f5253..4233710b65 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -1979,6 +1979,15 @@ def _finalize_cron_session(session_db, agent, job_id: str, job_name: str, cron_s logger.debug("Job '%s': session lifecycle classification failed: %s", job_id, e) try: _session_db.end_session(_final_cron_session_id, _end_reason) + # run_job owns cron-session finalization. AIAgent.close() also + # finalizes owned session rows by default; if we release the + # shared SessionDB first and then call agent.close(), that second + # end_session() reopens the just-closed SQLite handle (#94736). + # Once the scheduler has durably booked this terminal reason, + # disarm only the agent's redundant row-finalization step. Its + # remaining resource teardown still runs normally below. + if agent is not None: + agent._end_session_on_close = False except (Exception, KeyboardInterrupt) as e: logger.debug("Job '%s': failed to end session: %s", job_id, e) try: diff --git a/tests/cron/test_scheduler.py b/tests/cron/test_scheduler.py index 1cbb2870eb..ecbb19af52 100644 --- a/tests/cron/test_scheduler.py +++ b/tests/cron/test_scheduler.py @@ -587,6 +587,64 @@ class TestRunJobSessionPersistence: assert call_args[0][1] == "cron_complete" fake_db.close.assert_called_once() mock_agent.close.assert_called_once() + assert mock_agent._end_session_on_close is False + + + def test_run_job_disarms_agent_close_after_scheduler_finalizes_session(self, tmp_path): + """Cron owns the terminal session reason; agent.close must not end it twice. + + Regression for the #94736 teardown warning seen after every healthy cron + run: the scheduler closed its shared SessionDB, then AIAgent.close() tried + another end_session("agent_close"), forcing SessionDB to reopen solely + for a redundant write. + """ + job = {"id": "single-finalize", "name": "test", "prompt": "hello"} + fake_db = MagicMock() + fake_db.get_compression_tip.side_effect = lambda session_id: session_id + closed = False + calls_after_close = [] + + def close_db(): + nonlocal closed + closed = True + + def end_session(*args): + if closed: + calls_after_close.append(args) + + fake_db.close.side_effect = close_db + fake_db.end_session.side_effect = end_session + + with patch("cron.scheduler._hermes_home", tmp_path), \ + patch("cron.scheduler_delivery._resolve_origin", return_value=None), \ + patch("hermes_cli.env_loader.load_hermes_dotenv"), \ + patch("hermes_cli.env_loader.reset_secret_source_cache"), \ + patch("hermes_state_registry.acquire", return_value=fake_db), \ + patch( + "hermes_cli.runtime_provider.resolve_runtime_provider", + return_value={ + "api_key": "test-key", + "base_url": "https://example.invalid/v1", + "provider": "openrouter", + "api_mode": "chat_completions", + }, + ), \ + patch("run_agent.AIAgent") as mock_agent_cls: + mock_agent = MagicMock() + mock_agent.run_conversation.return_value = {"final_response": "ok"} + + def close_agent(): + if mock_agent._end_session_on_close: + fake_db.end_session(mock_agent.session_id, "agent_close") + + mock_agent.close.side_effect = close_agent + mock_agent_cls.return_value = mock_agent + success, *_ = run_job(job) + + assert success is True + assert fake_db.end_session.call_count == 1 + assert calls_after_close == [] + assert mock_agent._end_session_on_close is False @contextlib.contextmanager From 1256cf52fea2995fe16853e64dd07aa66111d3c4 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 19:39:27 +0530 Subject: [PATCH 145/276] chore: map RHODIZSECURITY contributor email for #102454 salvage --- contributors/emails/info.rhodiz@gmail.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/info.rhodiz@gmail.com diff --git a/contributors/emails/info.rhodiz@gmail.com b/contributors/emails/info.rhodiz@gmail.com new file mode 100644 index 0000000000..e2c2eeb516 --- /dev/null +++ b/contributors/emails/info.rhodiz@gmail.com @@ -0,0 +1,2 @@ +RHODIZSECURITY +# PR #102454 salvage From fa646582ae80eec22d280fdd74e8a99e055334e3 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:08:42 +0530 Subject: [PATCH 146/276] docs(cron): point the disarm comment at _teardown_cron_agent after the helper extraction --- cron/scheduler.py | 13 ++++++------- 1 file changed, 6 insertions(+), 7 deletions(-) diff --git a/cron/scheduler.py b/cron/scheduler.py index 4233710b65..c94562fe24 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -1979,13 +1979,12 @@ def _finalize_cron_session(session_db, agent, job_id: str, job_name: str, cron_s logger.debug("Job '%s': session lifecycle classification failed: %s", job_id, e) try: _session_db.end_session(_final_cron_session_id, _end_reason) - # run_job owns cron-session finalization. AIAgent.close() also - # finalizes owned session rows by default; if we release the - # shared SessionDB first and then call agent.close(), that second - # end_session() reopens the just-closed SQLite handle (#94736). - # Once the scheduler has durably booked this terminal reason, - # disarm only the agent's redundant row-finalization step. Its - # remaining resource teardown still runs normally below. + # The scheduler owns cron-session finalization. AIAgent.close() also + # finalizes owned session rows by default; once the shared SessionDB is + # released below, that second end_session() would reopen the just-closed + # SQLite handle (#94736). The reason is durably booked, so disarm only the + # agent's redundant row-finalization; its resource teardown still runs in + # _teardown_cron_agent. if agent is not None: agent._end_session_on_close = False except (Exception, KeyboardInterrupt) as e: From 1d039b47e0add3bd4c7c276e4b696464c2cceb17 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:41:21 +0530 Subject: [PATCH 147/276] test(cron): assert the behaviour (one end_session, none after release), not the flag --- tests/cron/test_scheduler.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/tests/cron/test_scheduler.py b/tests/cron/test_scheduler.py index ecbb19af52..21759419d3 100644 --- a/tests/cron/test_scheduler.py +++ b/tests/cron/test_scheduler.py @@ -587,7 +587,6 @@ class TestRunJobSessionPersistence: assert call_args[0][1] == "cron_complete" fake_db.close.assert_called_once() mock_agent.close.assert_called_once() - assert mock_agent._end_session_on_close is False def test_run_job_disarms_agent_close_after_scheduler_finalizes_session(self, tmp_path): @@ -644,7 +643,6 @@ class TestRunJobSessionPersistence: assert success is True assert fake_db.end_session.call_count == 1 assert calls_after_close == [] - assert mock_agent._end_session_on_close is False @contextlib.contextmanager From b4b6235239ac67a9dbaccb17a9062c6bfd158712 Mon Sep 17 00:00:00 2001 From: mengtanx Date: Sat, 5 Sep 2026 21:35:14 +0800 Subject: [PATCH 148/276] fix(update): credit launchd ai.hermes.gateway in fleet reconciliation (#103679) The restart phase records macOS LaunchAgent labels (ai.hermes.gateway). match_runtime_outcomes used a substring check for "hermes-gateway", so a successful Desktop update on the default profile always tripped "Planned runtimes the restart phase never touched" and exited 1. Use the exact systemd/launchd/s6 matcher for both plan reconciliation and abort-recovery so the two cannot drift. --- hermes_cli/update_cmd_fleet.py | 12 ++--- hermes_cli/update_inventory.py | 24 ++++++--- .../test_restart_plan_reconciliation.py | 49 +++++++++++++++++++ .../test_update_restart_recovery.py | 7 +++ 4 files changed, 77 insertions(+), 15 deletions(-) diff --git a/hermes_cli/update_cmd_fleet.py b/hermes_cli/update_cmd_fleet.py index d637db9556..aeaacecba2 100644 --- a/hermes_cli/update_cmd_fleet.py +++ b/hermes_cli/update_cmd_fleet.py @@ -15,6 +15,7 @@ from dataclasses import dataclass, field from pathlib import Path from hermes_cli.update_cmd_common import _best_effort +from hermes_cli.update_inventory import _gateway_service_matches_profile # Log-record parity with the origin module. logger = logging.getLogger("hermes_cli.update_cmd") @@ -542,15 +543,8 @@ def _surviving_gateway_pids_after_failed_restart(): return None -def _gateway_service_matches_profile(profile: str, service: object) -> bool: - """Match an exact gateway service/label (systemd/launchd/s6 shapes) to a profile. - - Never substring-match: ``foo`` must not claim ``hermes-gateway-foobar.service``. - """ - name = str(service).removesuffix(".service") - if profile == "default": - return name in {"hermes-gateway", "ai.hermes.gateway", "gateway", "gateway-default"} - return name in {f"hermes-gateway-{profile}", f"ai.hermes.gateway-{profile}", f"gateway-{profile}"} +# `_gateway_service_matches_profile` is imported from update_inventory so plan +# reconciliation and abort-recovery share one systemd/launchd/s6 matcher. _MANUAL_GATEWAY_SKIP_REASON = ( diff --git a/hermes_cli/update_inventory.py b/hermes_cli/update_inventory.py index 912d66dcc3..bc9d6bdd6c 100644 --- a/hermes_cli/update_inventory.py +++ b/hermes_cli/update_inventory.py @@ -288,13 +288,25 @@ def _serve_unit_matches_profile(profile: str, unit: object) -> bool: return name in {f"hermes-serve{suffix}", f"hermes-dashboard{suffix}"} +def _gateway_service_matches_profile(profile: str, service: object) -> bool: + """Match an exact gateway service/label (systemd/launchd/s6 shapes) to a profile. + + Never substring-match: ``foo`` must not claim ``hermes-gateway-foobar.service``. + Launchd labels are ``ai.hermes.gateway`` / ``ai.hermes.gateway-`` — they do + not contain the substring ``hermes-gateway``, so a successful macOS kickstart must + still credit the planned default gateway. A scope prefix (``user/hermes-gateway``, + ``gui/501/ai.hermes.gateway``) is stripped the same way serve units are. + """ + name = str(service).removesuffix(".service").rsplit("/", 1)[-1] + if profile == "default": + return name in {"hermes-gateway", "ai.hermes.gateway", "gateway", "gateway-default"} + return name in {f"hermes-gateway-{profile}", f"ai.hermes.gateway-{profile}", f"gateway-{profile}"} + + def _gateway_named_in(r: RuntimeRecord, names: set) -> bool: - # The bare "hermes-gateway" unit name is gateway-specific: a serve/dashboard runtime that merely - # shares the default profile is a different process the gateway restart never touched. - return any( - r.profile in name or (r.kind == "gateway" and r.profile == "default" and "hermes-gateway" in name) - for name in names - ) + # Gateway-only vocabulary: a serve/dashboard that merely shares the profile is a + # different process. Exact label match (systemd + launchd + s6), not substring. + return any(_gateway_service_matches_profile(r.profile, name) for name in names) def match_runtime_outcomes( diff --git a/tests/hermes_cli/test_restart_plan_reconciliation.py b/tests/hermes_cli/test_restart_plan_reconciliation.py index b87cc98f7a..91209886ae 100644 --- a/tests/hermes_cli/test_restart_plan_reconciliation.py +++ b/tests/hermes_cli/test_restart_plan_reconciliation.py @@ -133,6 +133,55 @@ def test_restarted_service_unit_matches_profile(): assert outcomes[0]["outcome"] == "restarted" +def test_launchd_default_gateway_restarted_via_ai_hermes_label(): + """macOS restart bookkeeping records ``ai.hermes.gateway``, which does not + contain the substring ``hermes-gateway``. The default-profile gateway must + still count as restarted — otherwise every Desktop update on launchd + exits 1 after a successful kickstart (receipt outcome=partial, tripwire + 'never touched').""" + outcomes = match_runtime_outcomes( + _plan(_rt("default", 400, supervisor="launchd")), + restarted_services=["ai.hermes.gateway"], relaunched_profiles=[], + externally_supervised_profiles=[], killed_pids=set(), failed_units=[], + ) + assert outcomes[0]["outcome"] == "restarted" + assert report_unaccounted_runtimes(outcomes) is False + + +def test_launchd_named_profile_and_failed_label(): + restarted = match_runtime_outcomes( + _plan(_rt("work", 401, supervisor="launchd")), + restarted_services=["ai.hermes.gateway-work"], relaunched_profiles=[], + externally_supervised_profiles=[], killed_pids=set(), failed_units=[], + ) + assert restarted[0]["outcome"] == "restarted" + failed = match_runtime_outcomes( + _plan(_rt("default", 402, supervisor="launchd")), + restarted_services=[], relaunched_profiles=[], + externally_supervised_profiles=[], killed_pids=set(), + failed_units=["ai.hermes.gateway"], + ) + assert failed[0]["outcome"] == "failed" + # Sibling label must not credit the default profile (exact match, not prefix). + sibling = match_runtime_outcomes( + _plan(_rt("default", 403, supervisor="launchd")), + restarted_services=["ai.hermes.gateway-foo"], relaunched_profiles=[], + externally_supervised_profiles=[], killed_pids=set(), failed_units=[], + ) + assert sibling[0]["outcome"] == "unaccounted" + + +def test_launchd_gateway_restart_does_not_credit_serve(): + """#100479 sibling: a launchd gateway restart is still gateway vocabulary.""" + outcomes = match_runtime_outcomes( + _plan(_rt("default", 100, supervisor="launchd"), _serve("default", 900)), + restarted_services=["ai.hermes.gateway"], relaunched_profiles=[], + externally_supervised_profiles=[], killed_pids=set(), failed_units=[], + ) + by_pid = {o["pid"]: o["outcome"] for o in outcomes} + assert by_pid == {100: "restarted", 900: "unaccounted"} + + def test_untouched_runtime_is_unaccounted_and_escalates(capsys): """The tripwire: plan saw it, NO bookkeeping mentions it.""" outcomes = match_runtime_outcomes( diff --git a/tests/hermes_cli/test_update_restart_recovery.py b/tests/hermes_cli/test_update_restart_recovery.py index 780cef66c4..d7c870c1b4 100644 --- a/tests/hermes_cli/test_update_restart_recovery.py +++ b/tests/hermes_cli/test_update_restart_recovery.py @@ -252,6 +252,13 @@ def test_service_matching_is_exact_for_overlapping_profile_names(): assert not update_cmd._gateway_service_matches_profile( "default", "ai.hermes.gateway-foo" ) + # Scope-qualified identities the restart phase may record. + assert update_cmd._gateway_service_matches_profile( + "default", "gui/501/ai.hermes.gateway" + ) + assert update_cmd._gateway_service_matches_profile( + "foo", "user/hermes-gateway-foo.service" + ) def test_recovery_child_restarts_each_profile_with_a_fresh_main(monkeypatch): From 11a333620a0a0c5ff761e81fe44e8d142894a935 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 19:39:27 +0530 Subject: [PATCH 149/276] chore: map mengtanx contributor email for #103680 salvage --- contributors/emails/mengtanx@gmail.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/mengtanx@gmail.com diff --git a/contributors/emails/mengtanx@gmail.com b/contributors/emails/mengtanx@gmail.com new file mode 100644 index 0000000000..af546afbec --- /dev/null +++ b/contributors/emails/mengtanx@gmail.com @@ -0,0 +1,2 @@ +mengtanx +# PR #103680 salvage From 80d636b7c5ac62b51a61e42b6fca84d6cf5fce6c Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:08:41 +0530 Subject: [PATCH 150/276] test(update): drop the launchd serve-isolation test already covered by the serve branch Serve/dashboard runtimes return from match_runtime_outcomes before _gateway_named_in is reached, and the systemd flavour of this contract is already pinned by test_serve_reconciles_against_its_own_unit_vocabulary. --- tests/hermes_cli/test_restart_plan_reconciliation.py | 11 ----------- 1 file changed, 11 deletions(-) diff --git a/tests/hermes_cli/test_restart_plan_reconciliation.py b/tests/hermes_cli/test_restart_plan_reconciliation.py index 91209886ae..ea25d915d8 100644 --- a/tests/hermes_cli/test_restart_plan_reconciliation.py +++ b/tests/hermes_cli/test_restart_plan_reconciliation.py @@ -171,17 +171,6 @@ def test_launchd_named_profile_and_failed_label(): assert sibling[0]["outcome"] == "unaccounted" -def test_launchd_gateway_restart_does_not_credit_serve(): - """#100479 sibling: a launchd gateway restart is still gateway vocabulary.""" - outcomes = match_runtime_outcomes( - _plan(_rt("default", 100, supervisor="launchd"), _serve("default", 900)), - restarted_services=["ai.hermes.gateway"], relaunched_profiles=[], - externally_supervised_profiles=[], killed_pids=set(), failed_units=[], - ) - by_pid = {o["pid"]: o["outcome"] for o in outcomes} - assert by_pid == {100: "restarted", 900: "unaccounted"} - - def test_untouched_runtime_is_unaccounted_and_escalates(capsys): """The tripwire: plan saw it, NO bookkeeping mentions it.""" outcomes = match_runtime_outcomes( From a262b2e372130fd59b013f7b10a7ed2ecb7f9357 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:41:16 +0530 Subject: [PATCH 151/276] refactor(update): drop the tombstone comment at the old matcher site and a duplicated exactness leg --- hermes_cli/update_cmd_fleet.py | 4 ---- tests/hermes_cli/test_restart_plan_reconciliation.py | 7 ------- 2 files changed, 11 deletions(-) diff --git a/hermes_cli/update_cmd_fleet.py b/hermes_cli/update_cmd_fleet.py index aeaacecba2..a4246b6a69 100644 --- a/hermes_cli/update_cmd_fleet.py +++ b/hermes_cli/update_cmd_fleet.py @@ -543,10 +543,6 @@ def _surviving_gateway_pids_after_failed_restart(): return None -# `_gateway_service_matches_profile` is imported from update_inventory so plan -# reconciliation and abort-recovery share one systemd/launchd/s6 matcher. - - _MANUAL_GATEWAY_SKIP_REASON = ( "manual gateway has no supervisor relaunch authority; left running for explicit operator restart" ) diff --git a/tests/hermes_cli/test_restart_plan_reconciliation.py b/tests/hermes_cli/test_restart_plan_reconciliation.py index ea25d915d8..4066d57c57 100644 --- a/tests/hermes_cli/test_restart_plan_reconciliation.py +++ b/tests/hermes_cli/test_restart_plan_reconciliation.py @@ -162,13 +162,6 @@ def test_launchd_named_profile_and_failed_label(): failed_units=["ai.hermes.gateway"], ) assert failed[0]["outcome"] == "failed" - # Sibling label must not credit the default profile (exact match, not prefix). - sibling = match_runtime_outcomes( - _plan(_rt("default", 403, supervisor="launchd")), - restarted_services=["ai.hermes.gateway-foo"], relaunched_profiles=[], - externally_supervised_profiles=[], killed_pids=set(), failed_units=[], - ) - assert sibling[0]["outcome"] == "unaccounted" def test_untouched_runtime_is_unaccounted_and_escalates(capsys): From dc4b88714bcb4b4347ad5027819cc9d9b3eb8530 Mon Sep 17 00:00:00 2001 From: cdepuy Date: Sat, 5 Sep 2026 11:23:34 -0700 Subject: [PATCH 152/276] fix(gateway): honor server retry_after for flood-capped sends instead of plain-text fallback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Flood-capped / rate-limited sends (Telegram FloodWait, Weixin bare RuntimeError) were not honored as retryable when retry_after was absent, so _send_with_retry fell through to the truncating plain-text fallback, which re-entered the server ban and dropped the message tail — silently burning the turn with no delivery. Three changes to _send_with_retry: 1. Route rate-limit checks through _is_rate_limited_error (single wrapper over classify_send_error) so rate limits are treated as transient even when the platform omits retry_after. 2. Reclassify rate-limit per retry attempt, not just the initial send, so the in-loop break/continue reflects the CURRENT attempt (covers transient->flood and rate-limited->formatting transitions). 3. Never take the truncating plain-text fallback for a rate-limited send — return the typed failure so the delivery ledger owns redelivery after the cooldown. And when retries exhaust on a rate-limited / retry_after-carrying failure, return BEFORE sending the delivery-failure notice: the notice send would land inside the same flood penalty and re-enter the ban ([0, 189, 378, 378] -> 3 sends, no 4th). Ordinary exhausted network errors keep the existing notice. Honors server retry_after (when present) as the backoff delay instead of the default exponential schedule. Rebased onto current main (was 5163 commits behind, CONFLICTING) and folded in reviewer feedback. Regression coverage in tests/gateway/test_send_retry.py: rate-limit classifier, rate-limited-without-retry_after retries & succeeds, rate-limited exhaustion returns a typed failure with no fallback and no notice inside the active flood penalty, and failure-kind transitions between attempts. --- gateway/platforms/base.py | 71 ++++++++++++++-- tests/gateway/test_send_retry.py | 136 +++++++++++++++++++++++++++++++ 2 files changed, 201 insertions(+), 6 deletions(-) diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index d7eb6dabeb..4d97849721 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -3156,6 +3156,16 @@ class BasePlatformAdapter(ABC): lowered = (error or "").lower() return any(pat in lowered for pat in _RETRYABLE_ERROR_PATTERNS) + @staticmethod + def _is_rate_limited_error(error: Optional[str]) -> bool: + """Return True if the error string classifies as a rate limit / flood cap. + + Single wrapper around :func:`classify_send_error` so the call sites in + :meth:`_send_with_retry` share one notion of "is this a rate limit" instead + of inline copies that could drift. + """ + return classify_send_error(None, error or "") == "rate_limited" + @staticmethod def _is_timeout_error(error: Optional[str]) -> bool: """Return True for read/write timeouts — NOT retryable and NOT a plain-text @@ -3221,7 +3231,19 @@ class BasePlatformAdapter(ABC): if result.success: return result error_str = result.error or "" - is_network = result.retryable or self._is_retryable_error(error_str) + # A rate-limited / flood-capped send is transient: it should back off + # (honoring the server's retry_after when present) rather than fall + # through to the plain-text fallback, which re-enters the ban and can + # truncate content. Gate on the platform-neutral classifier as well so + # platforms that surface a rate limit without a retry_after field + # (e.g. Weixin raising a bare RuntimeError) get the same treatment. + is_rate_limited = self._is_rate_limited_error(error_str) + is_network = ( + result.retryable + or is_rate_limited + or result.retry_after is not None + or self._is_retryable_error(error_str) + ) # Timeouts: not safe to retry (may have delivered) and not a formatting error. if not is_network and self._is_timeout_error(error_str): return result @@ -3242,10 +3264,36 @@ class BasePlatformAdapter(ABC): logger.info("[%s] Send succeeded on retry %d", self.name, attempt) return result error_str = result.error or "" - server_retry_after = result.retry_after # None unless the server asked again - if not (result.retryable or self._is_retryable_error(error_str)): - break # non-transient now — fall through to plain-text fallback - else: # retries exhausted — notify user + if result.retry_after is not None: + server_retry_after = result.retry_after + # The failure kind can change between attempts (a transient error may + # later surface as a flood/rate-limit, or a rate-limited send may give + # way to a permanent formatting error). Reclassify from the refreshed + # error_str on every attempt so the break/continue decision below + # reflects the current attempt, not a stale first-send classification. + is_rate_limited = self._is_rate_limited_error(error_str) + if not ( + result.retryable + or is_rate_limited + or result.retry_after is not None + or self._is_retryable_error(error_str) + ): + break # error switched to non-transient — fall through to plain-text fallback + else: + # All retries exhausted (loop completed without break) — notify user. + # If the final failure is a rate-limit / still carries a server + # retry_after, do NOT send the delivery-failure notice now: the notice + # send would land inside the same flood penalty and re-enter the ban + # (a fourth send at t=378 in an [0, 189, 378, 378] sequence). Return + # the typed failure so the delivery ledger owns redelivery after the + # cooldown instead — no extra sleep or request needed. + if self._is_rate_limited_error(error_str) or result.retry_after is not None: + logger.error( + "[%s] Rate-limited send exhausted retries; returning typed failure " + "for redelivery (no notice sent inside active flood penalty): %s", + self.name, error_str, + ) + return result logger.error("[%s] Failed to deliver response after %d retries: %s", self.name, max_retries, error_str) notice = ( "\u26a0\ufe0f Message delivery failed after multiple attempts. " @@ -3255,7 +3303,18 @@ class BasePlatformAdapter(ABC): except Exception as notify_err: logger.debug("[%s] Could not send delivery-failure notice: %s", self.name, notify_err) return result - # Non-network / post-retry formatting failure: try plain text as fallback + # Non-network / post-retry formatting failure: try plain text as fallback. + # Never attempt a truncating plain-text fallback for a rate-limited / + # flood-capped send: it re-enters the server ban and would drop the tail + # of the message. Return the typed failure so the delivery ledger owns + # redelivery after the cooldown instead. + if self._is_rate_limited_error(error_str): + logger.error( + "[%s] Rate-limited send not retried via plain-text fallback; " + "returning typed failure for redelivery: %s", + self.name, error_str, + ) + return result logger.warning("[%s] Send failed: %s — trying plain-text fallback", self.name, error_str) fallback_result = await _send(f"(Response formatting failed, plain text:)\n\n{content[:3500]}") if not fallback_result.success: diff --git a/tests/gateway/test_send_retry.py b/tests/gateway/test_send_retry.py index a3da168b3c..4e13581dfe 100644 --- a/tests/gateway/test_send_retry.py +++ b/tests/gateway/test_send_retry.py @@ -76,6 +76,32 @@ class TestIsTimeoutError: assert not _StubAdapter._is_timeout_error("") +# --------------------------------------------------------------------------- +# _is_rate_limited_error +# --------------------------------------------------------------------------- + +class TestIsRateLimitedError: + def test_none_is_not_rate_limited(self): + assert not _StubAdapter._is_rate_limited_error(None) + + def test_empty_is_not_rate_limited(self): + assert not _StubAdapter._is_rate_limited_error("") + + def test_flood_is_rate_limited(self): + assert _StubAdapter._is_rate_limited_error("flood control exceeded, retry in 60 seconds") + + def test_too_many_requests_is_rate_limited(self): + assert _StubAdapter._is_rate_limited_error("Too Many Requests: rate limit reached") + + def test_weixin_rate_limit_text_is_rate_limited(self): + assert _StubAdapter._is_rate_limited_error( + "iLink sendmessage rate limited; cooldown active for 42.0s" + ) + + def test_formatting_error_not_rate_limited(self): + assert not _StubAdapter._is_rate_limited_error("Bad Request: can't parse entities") + + # --------------------------------------------------------------------------- # _send_with_retry — success on first attempt # --------------------------------------------------------------------------- @@ -206,3 +232,113 @@ class TestSendWithRetryAfter: second_sleep = mock_sleep.call_args_list[1][0][0] assert second_sleep >= 29.0 # 30 - 1 (max jitter) + +# --------------------------------------------------------------------------- +# _send_with_retry — rate-limited sends (flood control) without retry_after +# --------------------------------------------------------------------------- +# Platforms like Weixin surface a rate limit as a bare error with NO retry_after +# field. These must still be treated as transient (back off / retry) rather than +# falling through to the truncating plain-text fallback, which would re-enter the +# server ban and drop the tail of the message. + +class TestSendWithRetryRateLimited: + @pytest.mark.asyncio + async def test_rate_limited_no_retry_after_backs_off_and_retries(self): + """A flood/rate-limit error WITHOUT retry_after still counts as network + (retryable) because classify_send_error == 'rate_limited', so it retries + and succeeds instead of falling through to plain-text fallback.""" + adapter = _StubAdapter() + adapter._send_results = [ + SendResult(success=False, error="flood control exceeded, timeout 30 seconds"), + SendResult(success=True, message_id="ok"), + ] + with patch("asyncio.sleep", new_callable=AsyncMock) as mock_sleep: + result = await adapter._send_with_retry("chat1", "hello", max_retries=2, base_delay=1.0) + # Must have slept (backoff), i.e. treated as transient — NOT immediate fallback + assert mock_sleep.called + assert result.success + assert len(adapter._send_calls) == 2 + # No plain-text fallback content on either call + assert all("plain text" not in c[1].lower() for c in adapter._send_calls) + + @pytest.mark.asyncio + async def test_rate_limited_exhausted_returns_typed_failure_not_fallback(self): + """When retries on a rate-limited send are exhausted, return the typed + failure for ledger redelivery. Never the truncating plain-text fallback, + and — because the final failure is still inside the flood penalty — never + a delivery-failure notice send that would re-enter the ban. 1 initial + 2 + retries = 3 sends, no fourth notice send.""" + adapter = _StubAdapter() + flood = SendResult(success=False, error="flood control exceeded, retry in 60 seconds") + adapter._send_results = [flood, flood, flood] + with patch("asyncio.sleep", new_callable=AsyncMock) as mock_sleep: + result = await adapter._send_with_retry("chat1", "hello", max_retries=2, base_delay=0) + assert not result.success + # 1 initial + 2 retries = 3 sends; NO delivery-failure notice (4th send) + # is sent inside the active flood penalty + assert len(adapter._send_calls) == 3 + # The only sends are the retries — no truncating "plain text" fallback and + # no notice triggered inside the flood penalty + for chat_id, content in adapter._send_calls: + assert "plain text" not in content.lower(), \ + f"rate-limited send must not fall through to plain-text fallback, got: {content[:40]!r}" + assert "delivery failed" not in content.lower() and \ + "Message delivery failed" not in content, \ + "no notice send inside active flood penalty" + + +# --------------------------------------------------------------------------- +# _send_with_retry — failure-kind transitions between attempts +# --------------------------------------------------------------------------- +# is_rate_limited is recomputed from the refreshed error_str on every retry, so +# the loop's continue/break decision reflects the CURRENT attempt, not the +# initial send's classification. These cover the two transitions that a stale +# classification would get wrong. + +class TestSendWithRetryFailureTypeTransitions: + @pytest.mark.asyncio + async def test_rate_limited_then_formatting_error_stops_retrying_and_falls_back(self): + """A rate-limited first attempt followed by a PERMANENT formatting error + on retry must stop retrying that permanent error and reach the existing + plain-text fallback — not keep retrying it (as a stale, still-true + is_rate_limited would).""" + adapter = _StubAdapter() + adapter._send_results = [ + SendResult(success=False, error="flood control exceeded, retry in 60 seconds"), + SendResult(success=False, error="Bad Request: can't parse entities"), + # fallback send (auto-succeeds via _next_result) + ] + with patch("asyncio.sleep", new_callable=AsyncMock) as mock_sleep: + result = await adapter._send_with_retry("chat1", "**bold**", max_retries=3, base_delay=0) + # The formatting error was not retried further: exactly 1 retry then the + # plain-text fallback. 1 initial + 1 retry + 1 fallback = 3 sends. + assert len(adapter._send_calls) == 3 + # The permanent formatting error switched attempts to non-transient: + # we fall through to (and return) the plain-text fallback. + assert "plain text" in adapter._send_calls[-1][1].lower() + assert result.success # fallback succeeded + # No delivery-failure notice was sent (this is a formatting fallback, not network exhaustion) + assert "delivery failed" not in adapter._send_calls[-1][1].lower() + + @pytest.mark.asyncio + async def test_transient_then_rate_limited_then_success_preserves_retry_budget(self): + """A transient network error followed by a rate-limited attempt must NOT + break the retry loop early: a later rate-limited result is still + transient, so the remaining retry budget is preserved and the send goes + on to succeed.""" + adapter = _StubAdapter() + adapter._send_results = [ + SendResult(success=False, error="httpx.ConnectError: connection refused"), + SendResult(success=False, error="flood control exceeded, retry in 30 seconds"), + SendResult(success=True, message_id="ok"), + ] + with patch("asyncio.sleep", new_callable=AsyncMock) as mock_sleep: + result = await adapter._send_with_retry("chat1", "hello", max_retries=3, base_delay=0) + # Initial + retry(rate-limited) + retry(success) = 3 sends, recovered. + assert result.success + assert len(adapter._send_calls) == 3 + # The rate-limited attempt was treated as transient (not a break), so no + # delivery-failure notice and no plain-text fallback. + assert all("plain text" not in c[1].lower() for c in adapter._send_calls) + assert "delivery failed" not in adapter._send_calls[-1][1].lower() + From 07ea974cc3b5c959ee3e9001edf980565f581da7 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 19:39:27 +0530 Subject: [PATCH 153/276] chore: map cdepuy contributor email for #100072 salvage --- contributors/emails/cdepuy@users.noreply.github.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/cdepuy@users.noreply.github.com diff --git a/contributors/emails/cdepuy@users.noreply.github.com b/contributors/emails/cdepuy@users.noreply.github.com new file mode 100644 index 0000000000..4f11612f43 --- /dev/null +++ b/contributors/emails/cdepuy@users.noreply.github.com @@ -0,0 +1,2 @@ +cdepuy +# PR #100072 salvage From dfffd0ee1ec750f0bb88f6d62f9583831f639ad0 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:22:44 +0530 Subject: [PATCH 154/276] fix(gateway): cap the inline server retry_after wait; return the typed failure past 60s Review follow-up on the salvaged #100072 flood handling. Honouring retry_after is right, but sleeping it verbatim in _send_with_retry would pin the send coroutine for a long penalty (a 97-minute FloodWait once froze inbound on every platform, #91969 -- the Telegram adapter already fails closed at 5s for the same reason and hands the wait to this loop). Past 60s the typed failure goes back to the delivery ledger, which owns redelivery after the cooldown. The six classifier tests collapse into one parametrized contract. --- gateway/platforms/base.py | 15 ++++++++++ tests/gateway/test_send_retry.py | 47 +++++++++++++++++++------------- 2 files changed, 43 insertions(+), 19 deletions(-) diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index 4d97849721..b16bb0e907 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -1660,6 +1660,11 @@ class SendResult: error_kind: Optional[str] = None +# Longest server ``retry_after`` ``_send_with_retry`` will sleep inline. Longer penalties return the +# typed failure so the delivery ledger owns the wait (#91969: a 97-minute FloodWait slept verbatim +# pinned the send coroutine and froze inbound on every platform). +_SEND_RETRY_INLINE_WAIT_CAP_SECS = 60.0 + # Platform-neutral send-failure kinds for ``SendResult.error_kind``: too_long (size cap), # bad_format (markup rejected; plain-text retry fixes), forbidden (the bot CANNOT reach the user), # not_found (chat/thread/message gone), rate_limited, transient (connection-level, retry-safe), @@ -3254,6 +3259,16 @@ class BasePlatformAdapter(ABC): backoff = server_retry_after if backoff is None: backoff = base_delay * (2 ** (attempt - 1)) + elif backoff > _SEND_RETRY_INLINE_WAIT_CAP_SECS: + # Never hold this coroutine open for a long server penalty: a 97-minute + # FloodWait slept verbatim once froze inbound on every platform (#91969). + # Return the typed failure; the delivery ledger redelivers after the cooldown. + logger.error( + "[%s] Server asked to retry after %.0fs (> %.0fs inline cap); returning " + "typed failure for redelivery instead of sleeping: %s", + self.name, backoff, _SEND_RETRY_INLINE_WAIT_CAP_SECS, error_str, + ) + return result delay = backoff + random.uniform(0, 1) server_retry_after = None logger.warning("[%s] Send failed (attempt %d/%d, retrying in %.1fs): %s", self.name, diff --git a/tests/gateway/test_send_retry.py b/tests/gateway/test_send_retry.py index 4e13581dfe..f893d0b3eb 100644 --- a/tests/gateway/test_send_retry.py +++ b/tests/gateway/test_send_retry.py @@ -81,25 +81,19 @@ class TestIsTimeoutError: # --------------------------------------------------------------------------- class TestIsRateLimitedError: - def test_none_is_not_rate_limited(self): - assert not _StubAdapter._is_rate_limited_error(None) - - def test_empty_is_not_rate_limited(self): - assert not _StubAdapter._is_rate_limited_error("") - - def test_flood_is_rate_limited(self): - assert _StubAdapter._is_rate_limited_error("flood control exceeded, retry in 60 seconds") - - def test_too_many_requests_is_rate_limited(self): - assert _StubAdapter._is_rate_limited_error("Too Many Requests: rate limit reached") - - def test_weixin_rate_limit_text_is_rate_limited(self): - assert _StubAdapter._is_rate_limited_error( - "iLink sendmessage rate limited; cooldown active for 42.0s" - ) - - def test_formatting_error_not_rate_limited(self): - assert not _StubAdapter._is_rate_limited_error("Bad Request: can't parse entities") + @pytest.mark.parametrize( + ("error", "rate_limited"), + [ + (None, False), + ("", False), + ("Bad Request: can't parse entities", False), + ("flood control exceeded, retry in 60 seconds", True), + # Platforms without a retry_after field (Weixin) surface a bare text. + ("iLink sendmessage rate limited; cooldown active for 42.0s", True), + ], + ) + def test_rate_limited_follows_the_shared_classifier(self, error, rate_limited): + assert _StubAdapter._is_rate_limited_error(error) is rate_limited # --------------------------------------------------------------------------- @@ -261,6 +255,21 @@ class TestSendWithRetryRateLimited: # No plain-text fallback content on either call assert all("plain text" not in c[1].lower() for c in adapter._send_calls) + @pytest.mark.asyncio + async def test_long_server_retry_after_returns_typed_failure_without_sleeping(self): + """A retry_after past the inline cap must not pin the coroutine (#91969): the + typed failure goes back to the delivery ledger, no sleep, no fallback, no notice.""" + adapter = _StubAdapter() + adapter._send_results = [ + SendResult(success=False, error="flood_control:5820", retry_after=5820.0), + ] + with patch("asyncio.sleep", new_callable=AsyncMock) as mock_sleep: + result = await adapter._send_with_retry("c", "hello") + assert result.success is False + assert result.retry_after == 5820.0 + mock_sleep.assert_not_called() + assert len(adapter._send_calls) == 1 # the original send only + @pytest.mark.asyncio async def test_rate_limited_exhausted_returns_typed_failure_not_fallback(self): """When retries on a rate-limited send are exhausted, return the typed From 9e4135fad89b01bc84c73a6ef405a2044f6b1f86 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:40:49 +0530 Subject: [PATCH 155/276] refactor(gateway): drop the unreachable post-loop rate-limit guard; trim tests to the contracts A rate-limited error classifies as network above and the retry loop only breaks on a non-transient, non-rate-limited error, so the fallback-path guard could never fire. Tests: the classifier delegator's vocabulary tests were a change-detector; two rate-limit loop tests were subsumed by the exhaustion contract (3 sends, no notice, no fallback). --- gateway/platforms/base.py | 15 ++------- tests/gateway/test_send_retry.py | 57 -------------------------------- 2 files changed, 3 insertions(+), 69 deletions(-) diff --git a/gateway/platforms/base.py b/gateway/platforms/base.py index b16bb0e907..e575aa7634 100644 --- a/gateway/platforms/base.py +++ b/gateway/platforms/base.py @@ -3318,18 +3318,9 @@ class BasePlatformAdapter(ABC): except Exception as notify_err: logger.debug("[%s] Could not send delivery-failure notice: %s", self.name, notify_err) return result - # Non-network / post-retry formatting failure: try plain text as fallback. - # Never attempt a truncating plain-text fallback for a rate-limited / - # flood-capped send: it re-enters the server ban and would drop the tail - # of the message. Return the typed failure so the delivery ledger owns - # redelivery after the cooldown instead. - if self._is_rate_limited_error(error_str): - logger.error( - "[%s] Rate-limited send not retried via plain-text fallback; " - "returning typed failure for redelivery: %s", - self.name, error_str, - ) - return result + # Non-network / post-retry formatting failure: try plain text as fallback. A + # rate-limited error never reaches here: it classifies as network above and the + # loop only breaks on a non-transient, non-rate-limited error. logger.warning("[%s] Send failed: %s — trying plain-text fallback", self.name, error_str) fallback_result = await _send(f"(Response formatting failed, plain text:)\n\n{content[:3500]}") if not fallback_result.success: diff --git a/tests/gateway/test_send_retry.py b/tests/gateway/test_send_retry.py index f893d0b3eb..011c878c05 100644 --- a/tests/gateway/test_send_retry.py +++ b/tests/gateway/test_send_retry.py @@ -80,22 +80,6 @@ class TestIsTimeoutError: # _is_rate_limited_error # --------------------------------------------------------------------------- -class TestIsRateLimitedError: - @pytest.mark.parametrize( - ("error", "rate_limited"), - [ - (None, False), - ("", False), - ("Bad Request: can't parse entities", False), - ("flood control exceeded, retry in 60 seconds", True), - # Platforms without a retry_after field (Weixin) surface a bare text. - ("iLink sendmessage rate limited; cooldown active for 42.0s", True), - ], - ) - def test_rate_limited_follows_the_shared_classifier(self, error, rate_limited): - assert _StubAdapter._is_rate_limited_error(error) is rate_limited - - # --------------------------------------------------------------------------- # _send_with_retry — success on first attempt # --------------------------------------------------------------------------- @@ -236,24 +220,6 @@ class TestSendWithRetryAfter: # server ban and drop the tail of the message. class TestSendWithRetryRateLimited: - @pytest.mark.asyncio - async def test_rate_limited_no_retry_after_backs_off_and_retries(self): - """A flood/rate-limit error WITHOUT retry_after still counts as network - (retryable) because classify_send_error == 'rate_limited', so it retries - and succeeds instead of falling through to plain-text fallback.""" - adapter = _StubAdapter() - adapter._send_results = [ - SendResult(success=False, error="flood control exceeded, timeout 30 seconds"), - SendResult(success=True, message_id="ok"), - ] - with patch("asyncio.sleep", new_callable=AsyncMock) as mock_sleep: - result = await adapter._send_with_retry("chat1", "hello", max_retries=2, base_delay=1.0) - # Must have slept (backoff), i.e. treated as transient — NOT immediate fallback - assert mock_sleep.called - assert result.success - assert len(adapter._send_calls) == 2 - # No plain-text fallback content on either call - assert all("plain text" not in c[1].lower() for c in adapter._send_calls) @pytest.mark.asyncio async def test_long_server_retry_after_returns_typed_failure_without_sleeping(self): @@ -328,26 +294,3 @@ class TestSendWithRetryFailureTypeTransitions: assert result.success # fallback succeeded # No delivery-failure notice was sent (this is a formatting fallback, not network exhaustion) assert "delivery failed" not in adapter._send_calls[-1][1].lower() - - @pytest.mark.asyncio - async def test_transient_then_rate_limited_then_success_preserves_retry_budget(self): - """A transient network error followed by a rate-limited attempt must NOT - break the retry loop early: a later rate-limited result is still - transient, so the remaining retry budget is preserved and the send goes - on to succeed.""" - adapter = _StubAdapter() - adapter._send_results = [ - SendResult(success=False, error="httpx.ConnectError: connection refused"), - SendResult(success=False, error="flood control exceeded, retry in 30 seconds"), - SendResult(success=True, message_id="ok"), - ] - with patch("asyncio.sleep", new_callable=AsyncMock) as mock_sleep: - result = await adapter._send_with_retry("chat1", "hello", max_retries=3, base_delay=0) - # Initial + retry(rate-limited) + retry(success) = 3 sends, recovered. - assert result.success - assert len(adapter._send_calls) == 3 - # The rate-limited attempt was treated as transient (not a break), so no - # delivery-failure notice and no plain-text fallback. - assert all("plain text" not in c[1].lower() for c in adapter._send_calls) - assert "delivery failed" not in adapter._send_calls[-1][1].lower() - From 5b6defc766be13e9250a3657aada366a8c01c114 Mon Sep 17 00:00:00 2001 From: ygd58 Date: Sat, 5 Sep 2026 09:38:32 +0000 Subject: [PATCH 156/276] fix(verify): refuse compose recipes against a project with running containers Fixes #103567. `hermes verify`'s compose recipe (`agent/verify/recipes.py::_detect_compose_recipe`) unconditionally runs `docker compose build` + `docker compose up` against any directory containing a `docker-compose.yml`, with no check for whether that directory is already a live deployment. On an image-hash change, `docker compose up` replaces the running containers -- destroying any state stored on a container-local path or anonymous volume (not a bind mount). This caused a real incident: several kanban workers each ran `hermes verify` as routine self-verification against a workspace that was simultaneously a live compose deployment; each invocation auto-built and auto-upped the project; the container carrying a live state DB was replaced (state lost, recovered from a periodic guard backup after a ~98 minute outage). Nothing in verify's output indicated it was about to touch live containers. Implemented the issue's "at minimum" suggested fix: before running any phase for a `kind == "compose"` recipe, check for running containers via `docker compose ps --status running` (read-only, never mutates anything). If any are found, refuse outright with a clear message naming the running containers, without ever invoking the real `build`/`start` commands as subprocesses. The refusal surfaces through the exact same PhaseResult/ VerifyResult contract every other phase failure uses (a synthetic failed "build" phase), so `hermes verify`'s existing JSON/text reporting requires no special-casing. The check is best-effort: when `docker compose ps` itself can't run (docker not installed, or some other environment issue), the function returns None (not an empty list) and verify proceeds normally rather than blocking on an unrelated environment gap -- matching how every other recipe kind already behaves when its own tooling is unavailable. The guard is scoped strictly to `kind == "compose"`; every other recipe kind (Node, Python, Go, Rust, Java, Makefile) is completely untouched and never even probes docker. Verified directly: constructed the exact reported scenario (a compose Recipe, `docker compose ps` returning running container names via a mocked subprocess.run) and confirmed run_verify() returns a failed result with a single synthetic phase, no build/start subprocess ever spawned. Also verified: no running containers proceeds normally (the real build phase is attempted); the docker-unavailable case proceeds normally rather than blocking; and a non-compose recipe never even calls the guard function. Added 5 regression tests to the existing tests/verify/test_environment_and_runner.py, following its established Recipe/run_verify/tmp_path test pattern. Verified as a genuine regression: reverted just the guard block and confirmed the primary refusal test fails (the guard's protective behavior is gone) -- a second test that would have gone on to actually invoke `docker compose build` as a real subprocess without the guard was interrupted rather than left to run against a sandbox with no docker daemon, since the first test's failure already conclusively demonstrates the regression. 24/24 pass in the modified test file; 21/21 pass across two additional related verify test files (no regression). --- agent/verify/runner.py | 43 ++++++++++- tests/verify/test_environment_and_runner.py | 85 +++++++++++++++++++++ 2 files changed, 127 insertions(+), 1 deletion(-) diff --git a/agent/verify/runner.py b/agent/verify/runner.py index eb6f746cfe..e9aed61934 100644 --- a/agent/verify/runner.py +++ b/agent/verify/runner.py @@ -182,6 +182,24 @@ def _run_start_phase( return ReadinessResult(url, ready, status, time.monotonic() - started, error, _tail(output)) +def _running_compose_containers(root: Path) -> list[str] | None: + """Names of currently-running containers for the compose project at *root*, + via ``docker compose ps`` (read-only; never mutates anything). ``None`` when the + check itself could not run (docker/compose unavailable, or not a compose + project here) -- callers must not treat that as "no running containers". + """ + try: + result = subprocess.run( + ["docker", "compose", "ps", "--status", "running", "--format", "{{.Name}}"], + cwd=root, capture_output=True, text=True, timeout=15, stdin=subprocess.DEVNULL, + ) + except (OSError, subprocess.TimeoutExpired): + return None + if result.returncode != 0: + return None + return [line for line in result.stdout.splitlines() if line.strip()] + + def run_verify( root: Path, recipe: Recipe, phases: tuple[str, ...] | list[str] | None = None, phase_timeout: float = DEFAULT_PHASE_TIMEOUT, ready_timeout: float = DEFAULT_READY_TIMEOUT, @@ -189,11 +207,34 @@ def run_verify( on_output: Callable[[str], None] | None = None, ) -> VerifyResult: """Run the selected command phases sequentially, then (unless ``skip_start`` or a - phase failed) boot ``recipe.start``, poll readiness, and tear the process group down.""" + phase failed) boot ``recipe.start``, poll readiness, and tear the process group down. + + A ``compose`` recipe refuses outright when the project already has running + containers: ``docker compose build`` + ``up`` replaces them on an image-hash + change, destroying any container-local state they carry -- this has caused a + real state-loss incident (#103567). The check is best-effort and read-only + (``docker compose ps``); when it cannot run at all, verify proceeds rather than + blocking on an unrelated environment gap, matching every other recipe kind's + behavior when its own tooling is unavailable.""" root = Path(root) selected = tuple(phases) if phases else PHASE_ORDER + ("start",) result = VerifyResult(recipe_name=recipe.name) + if recipe.kind == "compose": + running = _running_compose_containers(root) + if running: + result.phases.append(PhaseResult( + phase="build", command=recipe.build[0] if recipe.build else "docker compose build", + exit_code=1, duration=0.0, output_tail=( + "Refusing to run: this compose project already has running " + f"container(s) ({', '.join(running)}). `docker compose build` + " + "`up` would replace them on an image-hash change, destroying any " + "container-local state they carry. If you intend to rebuild this " + "live deployment, run `docker compose build`/`up` yourself." + ), + )) + return result + for phase in PHASE_ORDER: if phase not in selected: continue diff --git a/tests/verify/test_environment_and_runner.py b/tests/verify/test_environment_and_runner.py index 4c1d888562..31f0cc8307 100644 --- a/tests/verify/test_environment_and_runner.py +++ b/tests/verify/test_environment_and_runner.py @@ -4,6 +4,7 @@ import http.server import json import threading import time +from unittest.mock import MagicMock, patch from agent.verify.environment import ( load_manifest, @@ -110,6 +111,90 @@ class TestRunner: result = run_verify(tmp_path, recipe, skip_start=True) assert result.ok + +class TestComposeGuard: + """Regression for issue #103567: a compose recipe must refuse to run + against a project directory that's already a live compose deployment + with running containers -- ``docker compose build`` + ``up`` replaces + them on an image-hash change, destroying container-local state (a real + incident: a kanban worker's state DB was lost this way).""" + + def _compose_recipe(self): + return Recipe( + name="docker-compose project", kind="compose", + build=["docker compose build"], start="docker compose up", + evidence=["Detected docker-compose.yml"], + ) + + def test_refuses_when_containers_are_running(self, tmp_path, monkeypatch): + fake_ps = MagicMock(returncode=0, stdout="myproject-db-1\nmyproject-web-1\n") + monkeypatch.setattr("subprocess.run", MagicMock(return_value=fake_ps)) + + result = run_verify(tmp_path, self._compose_recipe(), skip_start=True) + + assert not result.ok + assert len(result.phases) == 1 + assert "myproject-db-1" in result.phases[0].output_tail + assert "myproject-web-1" in result.phases[0].output_tail + assert "Refusing to run" in result.phases[0].output_tail + + def test_does_not_actually_invoke_build_or_start_when_refusing(self, tmp_path, monkeypatch): + """The refusal must be a synthetic result -- the real build/start + commands (which would trigger the destructive replace) must never + actually be spawned.""" + fake_ps = MagicMock(returncode=0, stdout="myproject-db-1\n") + calls: list[list[str] | str] = [] + + def tracking_run(cmd, *args, **kwargs): + calls.append(cmd) + return fake_ps + + monkeypatch.setattr("subprocess.run", tracking_run) + + run_verify(tmp_path, self._compose_recipe(), skip_start=False) + + # Only the read-only "docker compose ps" probe ran -- never + # "docker compose build" or "docker compose up" as an actual subprocess. + assert len(calls) == 1 + assert calls[0][:3] == ["docker", "compose", "ps"] + + def test_proceeds_normally_with_no_running_containers(self, tmp_path, monkeypatch): + fake_ps = MagicMock(returncode=0, stdout="") + monkeypatch.setattr("subprocess.run", MagicMock(return_value=fake_ps)) + + with patch("agent.verify.runner._run_phase_command") as mock_phase: + mock_phase.return_value = MagicMock(ok=True, phase="build") + result = run_verify(tmp_path, self._compose_recipe(), phases=("build",)) + + assert mock_phase.called + + def test_proceeds_when_the_docker_probe_itself_is_unavailable(self, tmp_path, monkeypatch): + """docker/compose not installed, or this isn't actually a compose + project's directory from docker's perspective -- must not block + verify on an unrelated environment gap.""" + monkeypatch.setattr( + "subprocess.run", + MagicMock(side_effect=FileNotFoundError("docker not found")), + ) + + with patch("agent.verify.runner._run_phase_command") as mock_phase: + mock_phase.return_value = MagicMock(ok=True, phase="build") + result = run_verify(tmp_path, self._compose_recipe(), phases=("build",)) + + assert mock_phase.called + + def test_non_compose_recipes_are_unaffected(self, tmp_path, monkeypatch): + """The guard is scoped to kind == "compose" only -- a Node/Python/etc. + recipe must never even check for running compose containers.""" + probe = MagicMock() + monkeypatch.setattr("agent.verify.runner._running_compose_containers", probe) + recipe = Recipe(name="x", kind="node", build=["true"]) + + result = run_verify(tmp_path, recipe, skip_start=True) + + assert result.ok + probe.assert_not_called() + def test_result_to_dict(self, tmp_path): recipe = Recipe(name="x", test=["true"]) payload = run_verify(tmp_path, recipe, skip_start=True).to_dict() From 6ccf485f37522adbe1b50074b49b3b942cef7422 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:07:54 +0530 Subject: [PATCH 157/276] fix(verify): refuse compose build/up when the live-container probe cannot answer Review follow-up on the salvaged #103577 guard. Fail-open on a hung or failing `docker compose ps` reopened the #103567 window: containers may be live and unobservable, and `docker compose build` would fail against the same daemon anyway, so refusing loses nothing. Only a missing docker binary falls through. The guard now also runs only when a mutating phase (build, or start without skip_start) is selected. Tests collapsed to three contracts: refusal spawns nothing mutating (running / timed out / failed probe), fall-through (none running / docker absent), guard skipped for non-mutating phase selections. --- agent/verify/runner.py | 20 +++-- tests/verify/test_environment_and_runner.py | 81 ++++++++++----------- 2 files changed, 52 insertions(+), 49 deletions(-) diff --git a/agent/verify/runner.py b/agent/verify/runner.py index e9aed61934..c327dde636 100644 --- a/agent/verify/runner.py +++ b/agent/verify/runner.py @@ -184,19 +184,26 @@ def _run_start_phase( def _running_compose_containers(root: Path) -> list[str] | None: """Names of currently-running containers for the compose project at *root*, - via ``docker compose ps`` (read-only; never mutates anything). ``None`` when the - check itself could not run (docker/compose unavailable, or not a compose - project here) -- callers must not treat that as "no running containers". + via ``docker compose ps`` (read-only; never mutates anything). + + ``None`` only when docker itself is absent -- the build phase would fail the + same way, so there is nothing to protect. A hung daemon or a non-zero probe + is reported as a refusal: containers may be live and unobservable, which is + exactly the #103567 loss window, and ``docker compose build`` would not have + fared better against the same daemon. """ try: result = subprocess.run( ["docker", "compose", "ps", "--status", "running", "--format", "{{.Name}}"], cwd=root, capture_output=True, text=True, timeout=15, stdin=subprocess.DEVNULL, ) - except (OSError, subprocess.TimeoutExpired): + except FileNotFoundError: return None + except subprocess.TimeoutExpired: + return [""] if result.returncode != 0: - return None + detail = (result.stderr or result.stdout or "").strip().splitlines() + return [f""] return [line for line in result.stdout.splitlines() if line.strip()] @@ -220,7 +227,8 @@ def run_verify( selected = tuple(phases) if phases else PHASE_ORDER + ("start",) result = VerifyResult(recipe_name=recipe.name) - if recipe.kind == "compose": + mutating = ("build" in selected) or ("start" in selected and not skip_start) + if recipe.kind == "compose" and mutating: running = _running_compose_containers(root) if running: result.phases.append(PhaseResult( diff --git a/tests/verify/test_environment_and_runner.py b/tests/verify/test_environment_and_runner.py index 31f0cc8307..2108a2fe36 100644 --- a/tests/verify/test_environment_and_runner.py +++ b/tests/verify/test_environment_and_runner.py @@ -2,10 +2,13 @@ import http.server import json +import subprocess import threading import time from unittest.mock import MagicMock, patch +import pytest + from agent.verify.environment import ( load_manifest, load_or_detect, @@ -126,73 +129,65 @@ class TestComposeGuard: evidence=["Detected docker-compose.yml"], ) - def test_refuses_when_containers_are_running(self, tmp_path, monkeypatch): - fake_ps = MagicMock(returncode=0, stdout="myproject-db-1\nmyproject-web-1\n") - monkeypatch.setattr("subprocess.run", MagicMock(return_value=fake_ps)) - - result = run_verify(tmp_path, self._compose_recipe(), skip_start=True) - - assert not result.ok - assert len(result.phases) == 1 - assert "myproject-db-1" in result.phases[0].output_tail - assert "myproject-web-1" in result.phases[0].output_tail - assert "Refusing to run" in result.phases[0].output_tail - - def test_does_not_actually_invoke_build_or_start_when_refusing(self, tmp_path, monkeypatch): - """The refusal must be a synthetic result -- the real build/start - commands (which would trigger the destructive replace) must never - actually be spawned.""" - fake_ps = MagicMock(returncode=0, stdout="myproject-db-1\n") + @pytest.mark.parametrize( + "probe", + [ + pytest.param(MagicMock(returncode=0, stdout="myproject-db-1\nmyproject-web-1\n"), id="containers-running"), + pytest.param(subprocess.TimeoutExpired(cmd="docker", timeout=15), id="probe-timed-out"), + pytest.param(MagicMock(returncode=1, stdout="", stderr="permission denied on docker.sock"), id="probe-failed"), + ], + ) + def test_refuses_and_spawns_nothing_mutating_when_live_state_cannot_be_ruled_out( + self, tmp_path, monkeypatch, probe + ): calls: list[list[str] | str] = [] def tracking_run(cmd, *args, **kwargs): calls.append(cmd) - return fake_ps + if isinstance(probe, BaseException): + raise probe + return probe monkeypatch.setattr("subprocess.run", tracking_run) - run_verify(tmp_path, self._compose_recipe(), skip_start=False) + result = run_verify(tmp_path, self._compose_recipe(), skip_start=False) - # Only the read-only "docker compose ps" probe ran -- never - # "docker compose build" or "docker compose up" as an actual subprocess. + assert not result.ok + assert "Refusing to run" in result.phases[0].output_tail + if isinstance(probe, MagicMock) and probe.returncode == 0: + assert "myproject-db-1" in result.phases[0].output_tail + assert "myproject-web-1" in result.phases[0].output_tail + # Only the read-only probe ran -- never build or up. assert len(calls) == 1 assert calls[0][:3] == ["docker", "compose", "ps"] - def test_proceeds_normally_with_no_running_containers(self, tmp_path, monkeypatch): - fake_ps = MagicMock(returncode=0, stdout="") - monkeypatch.setattr("subprocess.run", MagicMock(return_value=fake_ps)) - - with patch("agent.verify.runner._run_phase_command") as mock_phase: - mock_phase.return_value = MagicMock(ok=True, phase="build") - result = run_verify(tmp_path, self._compose_recipe(), phases=("build",)) - - assert mock_phase.called - - def test_proceeds_when_the_docker_probe_itself_is_unavailable(self, tmp_path, monkeypatch): - """docker/compose not installed, or this isn't actually a compose - project's directory from docker's perspective -- must not block - verify on an unrelated environment gap.""" + @pytest.mark.parametrize( + "probe", + [ + pytest.param(MagicMock(returncode=0, stdout=""), id="none-running"), + pytest.param(FileNotFoundError("docker not found"), id="docker-absent"), + ], + ) + def test_proceeds_when_no_live_containers_or_docker_is_absent(self, tmp_path, monkeypatch, probe): monkeypatch.setattr( "subprocess.run", - MagicMock(side_effect=FileNotFoundError("docker not found")), + MagicMock(side_effect=probe) if isinstance(probe, BaseException) else MagicMock(return_value=probe), ) with patch("agent.verify.runner._run_phase_command") as mock_phase: mock_phase.return_value = MagicMock(ok=True, phase="build") - result = run_verify(tmp_path, self._compose_recipe(), phases=("build",)) + run_verify(tmp_path, self._compose_recipe(), phases=("build",)) assert mock_phase.called - def test_non_compose_recipes_are_unaffected(self, tmp_path, monkeypatch): - """The guard is scoped to kind == "compose" only -- a Node/Python/etc. - recipe must never even check for running compose containers.""" + def test_guard_skipped_when_no_mutating_phase_selected(self, tmp_path, monkeypatch): probe = MagicMock() monkeypatch.setattr("agent.verify.runner._running_compose_containers", probe) - recipe = Recipe(name="x", kind="node", build=["true"]) - result = run_verify(tmp_path, recipe, skip_start=True) + with patch("agent.verify.runner._run_phase_command") as mock_phase: + mock_phase.return_value = MagicMock(ok=True, phase="test") + run_verify(tmp_path, Recipe(name="x", kind="compose", test=["true"]), phases=("test",)) - assert result.ok probe.assert_not_called() def test_result_to_dict(self, tmp_path): From 00140a85574d4fc9e42f977a167aeda899b50ca9 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:39:50 +0530 Subject: [PATCH 158/276] refactor(verify): one honest return type for the compose live-state probe (str | None) The helper returned container names OR a synthetic pseudo-name so the caller would refuse; on a hung daemon that rendered as "already has running container(s) ()". It now returns the refusal reason or None, and the caller prints the reason as given. --- agent/verify/runner.py | 34 ++++++++++----------- tests/verify/test_environment_and_runner.py | 13 ++------ 2 files changed, 19 insertions(+), 28 deletions(-) diff --git a/agent/verify/runner.py b/agent/verify/runner.py index c327dde636..4016c1638b 100644 --- a/agent/verify/runner.py +++ b/agent/verify/runner.py @@ -182,15 +182,13 @@ def _run_start_phase( return ReadinessResult(url, ready, status, time.monotonic() - started, error, _tail(output)) -def _running_compose_containers(root: Path) -> list[str] | None: - """Names of currently-running containers for the compose project at *root*, - via ``docker compose ps`` (read-only; never mutates anything). +def _compose_live_state_reason(root: Path) -> str | None: + """Why ``docker compose build``/``up`` must not run at *root*, or ``None`` to proceed. - ``None`` only when docker itself is absent -- the build phase would fail the - same way, so there is nothing to protect. A hung daemon or a non-zero probe - is reported as a refusal: containers may be live and unobservable, which is - exactly the #103567 loss window, and ``docker compose build`` would not have - fared better against the same daemon. + Read-only ``docker compose ps`` probe. Only a missing docker binary proceeds -- the + build phase would fail the same way, so there is nothing to protect. A hung daemon + or a non-zero probe refuses: containers may be live and unobservable, which is + exactly the #103567 loss window. """ try: result = subprocess.run( @@ -200,11 +198,12 @@ def _running_compose_containers(root: Path) -> list[str] | None: except FileNotFoundError: return None except subprocess.TimeoutExpired: - return [""] + return "docker compose ps timed out after 15s; live containers cannot be ruled out" if result.returncode != 0: detail = (result.stderr or result.stdout or "").strip().splitlines() - return [f""] - return [line for line in result.stdout.splitlines() if line.strip()] + return f"docker compose ps failed (exit {result.returncode}): {detail[-1] if detail else 'no output'}" + names = [line for line in result.stdout.splitlines() if line.strip()] + return f"this compose project already has running container(s): {', '.join(names)}" if names else None def run_verify( @@ -229,16 +228,15 @@ def run_verify( mutating = ("build" in selected) or ("start" in selected and not skip_start) if recipe.kind == "compose" and mutating: - running = _running_compose_containers(root) - if running: + reason = _compose_live_state_reason(root) + if reason: result.phases.append(PhaseResult( phase="build", command=recipe.build[0] if recipe.build else "docker compose build", exit_code=1, duration=0.0, output_tail=( - "Refusing to run: this compose project already has running " - f"container(s) ({', '.join(running)}). `docker compose build` + " - "`up` would replace them on an image-hash change, destroying any " - "container-local state they carry. If you intend to rebuild this " - "live deployment, run `docker compose build`/`up` yourself." + f"Refusing to run: {reason}. `docker compose build` + `up` would replace " + "live containers on an image-hash change, destroying any container-local " + "state they carry. If you intend to rebuild this live deployment, run " + "`docker compose build`/`up` yourself." ), )) return result diff --git a/tests/verify/test_environment_and_runner.py b/tests/verify/test_environment_and_runner.py index 2108a2fe36..89905d14a8 100644 --- a/tests/verify/test_environment_and_runner.py +++ b/tests/verify/test_environment_and_runner.py @@ -116,11 +116,7 @@ class TestRunner: class TestComposeGuard: - """Regression for issue #103567: a compose recipe must refuse to run - against a project directory that's already a live compose deployment - with running containers -- ``docker compose build`` + ``up`` replaces - them on an image-hash change, destroying container-local state (a real - incident: a kanban worker's state DB was lost this way).""" + """#103567: a compose recipe must not build/up over a live deployment.""" def _compose_recipe(self): return Recipe( @@ -153,10 +149,7 @@ class TestComposeGuard: result = run_verify(tmp_path, self._compose_recipe(), skip_start=False) assert not result.ok - assert "Refusing to run" in result.phases[0].output_tail - if isinstance(probe, MagicMock) and probe.returncode == 0: - assert "myproject-db-1" in result.phases[0].output_tail - assert "myproject-web-1" in result.phases[0].output_tail + assert result.phases[0].exit_code == 1 # Only the read-only probe ran -- never build or up. assert len(calls) == 1 assert calls[0][:3] == ["docker", "compose", "ps"] @@ -182,7 +175,7 @@ class TestComposeGuard: def test_guard_skipped_when_no_mutating_phase_selected(self, tmp_path, monkeypatch): probe = MagicMock() - monkeypatch.setattr("agent.verify.runner._running_compose_containers", probe) + monkeypatch.setattr("agent.verify.runner._compose_live_state_reason", probe) with patch("agent.verify.runner._run_phase_command") as mock_phase: mock_phase.return_value = MagicMock(ok=True, phase="test") From 25be5e212054fb6467628ef45b63d025802130b5 Mon Sep 17 00:00:00 2001 From: Rob Aleck <1039058+mnbf9rca@users.noreply.github.com> Date: Sun, 23 Aug 2026 21:58:09 +0800 Subject: [PATCH 159/276] fix(email): send IMAP ID only when the server advertises the capability MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _send_imap_id() sent RFC 2971 ID unconditionally after login (a 163/NetEase requirement). Servers without the extension can react badly: Purelymail answers with an untagged '* BYE Unknown command.' and closes the connection, which imaplib cannot surface at the call site — the failure appears one command later as 'IMAP connection failed: command: SELECT => Unknown command.' and the adapter retries forever. Gate the ID send on the server's advertised capabilities (imap.capabilities, populated by imaplib at connect), keeping the existing exception handler for servers that advertise ID but reject it. Ports PR #39861 to the current plugin layout, as requested by the hermes-sweeper review there. Fixes #39856 Co-authored-by: liuhao1024 --- plugins/platforms/email/adapter.py | 14 +++++++- tests/gateway/test_email.py | 51 ++++++++++++++++++++++++++++++ 2 files changed, 64 insertions(+), 1 deletion(-) diff --git a/plugins/platforms/email/adapter.py b/plugins/platforms/email/adapter.py index 5f68843c41..efcc623897 100644 --- a/plugins/platforms/email/adapter.py +++ b/plugins/platforms/email/adapter.py @@ -141,7 +141,19 @@ def _open_smtp(host: str, port: int, security: str, ctx: ssl.SSLContext, smtp_cl def _send_imap_id(imap: "imaplib.IMAP4") -> None: """Send RFC 2971 IMAP ID: 163/NetEase require it after LOGIN (else every UID command - returns ``BYE Unsafe Login``); other servers may reject it, so failures are swallowed.""" + returns ``BYE Unsafe Login``); other servers may reject it, so failures are swallowed. + + Sent only when the server advertises ``ID`` (RFC 2971 requires advertising it): a server + without the extension can answer an untagged ``* BYE Unknown command.`` and close the + connection, which imaplib cannot surface here — the failure appears one command later + as a misleading SELECT error and the adapter retries forever (Purelymail, #39856). + ``imap.capabilities`` is populated by imaplib at connect, so the check is free.""" + caps = getattr(imap, "capabilities", ()) or () + if "ID" not in caps: + logger.debug( + "[Email] Server does not advertise IMAP ID capability; skipping ID" + ) + return try: try: from hermes_cli import __version__ as _hermes_version diff --git a/tests/gateway/test_email.py b/tests/gateway/test_email.py index e0ea57ce3b..07278a7b84 100644 --- a/tests/gateway/test_email.py +++ b/tests/gateway/test_email.py @@ -939,6 +939,7 @@ class TestImapIdExtensionForNetEase(unittest.TestCase): adapter = self._make_adapter() mock_imap = MagicMock() + mock_imap.capabilities = ("IMAP4REV1", "ID", "UIDPLUS") mock_imap.uid.return_value = ("OK", [b""]) with patch("imaplib.IMAP4_SSL", return_value=mock_imap), \ @@ -962,6 +963,56 @@ class TestImapIdExtensionForNetEase(unittest.TestCase): self.assertIn("login", names) self.assertLess(names.index("login"), names.index("xatom")) + def test_send_imap_id_sent_when_capability_advertised(self): + """ID goes out when the server lists it in CAPABILITY.""" + from plugins.platforms.email.adapter import _send_imap_id + + mock_imap = MagicMock() + mock_imap.capabilities = ("IMAP4REV1", "ID", "UIDPLUS") + + _send_imap_id(mock_imap) + + mock_imap.xatom.assert_called_once() + self.assertEqual(mock_imap.xatom.call_args.args[0], "ID") + + def test_send_imap_id_skipped_when_capability_absent(self): + """Servers that do not advertise ID must not receive it: Purelymail + answers the unknown command with an untagged ``* BYE Unknown + command.`` and drops the connection, which imaplib surfaces one + command later as a misleading SELECT failure.""" + from plugins.platforms.email.adapter import _send_imap_id + + mock_imap = MagicMock() + mock_imap.capabilities = ("IMAP4REV1", "UIDPLUS") + + _send_imap_id(mock_imap) + + mock_imap.xatom.assert_not_called() + + def test_send_imap_id_skipped_when_capabilities_attribute_missing(self): + """A connection object without a capabilities attribute must not + crash the helper; fail toward not sending optional commands.""" + from plugins.platforms.email.adapter import _send_imap_id + + mock_imap = MagicMock(spec=["xatom"]) + + _send_imap_id(mock_imap) + + mock_imap.xatom.assert_not_called() + + def test_send_imap_id_rejection_still_swallowed(self): + """A server that advertises ID but rejects it with a tagged error + keeps the existing best-effort handling: no exception escapes.""" + from plugins.platforms.email.adapter import _send_imap_id + + mock_imap = MagicMock() + mock_imap.capabilities = ("IMAP4REV1", "ID") + mock_imap.xatom.side_effect = Exception("BAD ID rejected") + + _send_imap_id(mock_imap) # must not raise + + mock_imap.xatom.assert_called_once() + class TestConnectSmtp(unittest.TestCase): """Test _connect_smtp() helper: protocol selection and IPv6 fallback.""" From 14ca27fa0601144b6ea4af1408a3b37c22456072 Mon Sep 17 00:00:00 2001 From: Ken Watts <8055558+mechkw@users.noreply.github.com> Date: Sat, 5 Sep 2026 16:21:06 +0000 Subject: [PATCH 160/276] test(email): verify IMAP ID negotiation over the wire Exercise startup and polling against a local IMAP peer using real imaplib. Cover missing ID, accepted ID, and tagged ID rejection without external accounts or network dependencies. Supports NousResearch/hermes-agent#92979 and #39856. --- tests/gateway/test_email_imap_id_protocol.py | 124 +++++++++++++++++++ 1 file changed, 124 insertions(+) create mode 100644 tests/gateway/test_email_imap_id_protocol.py diff --git a/tests/gateway/test_email_imap_id_protocol.py b/tests/gateway/test_email_imap_id_protocol.py new file mode 100644 index 0000000000..c44e9c9dbb --- /dev/null +++ b/tests/gateway/test_email_imap_id_protocol.py @@ -0,0 +1,124 @@ +"""Exercise IMAP ID negotiation with real imaplib and a local protocol peer. + +No external mail service or credentials are used. Unsupported ID gets BAD +followed by BYE: swallowing the ID error still leaves SELECT unable to proceed. +""" + +import asyncio +import imaplib +import socketserver +import threading +from contextlib import contextmanager +from unittest.mock import MagicMock + +import pytest + + +@contextmanager +def imap_peer(id_mode): + commands = [] + + class Handler(socketserver.StreamRequestHandler): + def handle(self): + self.connection.settimeout(5) + try: + self.exchange() + except (TimeoutError, ConnectionError): + # Bound cleanup even if the client fails before LOGOUT. + return + + def exchange(self): + self.wfile.write(b"* OK test IMAP ready\r\n") + while raw := self.rfile.readline(): + tag, command, *_ = raw.decode("ascii").strip().split() + command = command.upper() + commands.append(command) + prefix = tag.encode("ascii") + if command == "CAPABILITY": + caps = b"IMAP4rev1" + (b" ID" if id_mode != "absent" else b"") + self.wfile.write(b"* CAPABILITY " + caps + b"\r\n") + elif command == "ID": + if id_mode == "absent": + self.wfile.write( + prefix + + b" BAD ID unsupported\r\n* BYE Unknown command.\r\n" + ) + return + if id_mode == "reject": + self.wfile.write(prefix + b" BAD ID rejected\r\n") + continue + self.wfile.write(b"* ID NIL\r\n") + elif command == "SELECT": + self.wfile.write(b"* 0 EXISTS\r\n") + elif command == "UID": + self.wfile.write(b"* SEARCH\r\n") + elif command == "LOGOUT": + self.wfile.write( + b"* BYE logging out\r\n" + prefix + b" OK LOGOUT\r\n" + ) + return + elif command != "LOGIN": + self.wfile.write(prefix + b" BAD unexpected command\r\n") + continue + self.wfile.write(prefix + b" OK completed\r\n") + + # Non-daemon handler threads are joined by server_close on context exit; + # socket timeouts also bound teardown when the client fails prematurely. + with socketserver.ThreadingTCPServer(("127.0.0.1", 0), Handler) as server: + thread = threading.Thread( + target=server.serve_forever, + kwargs={"poll_interval": 0.05}, + daemon=True, + ) + thread.start() + try: + yield ("127.0.0.1", int(server.server_address[1])), commands + finally: + server.shutdown() + thread.join(timeout=5) + assert not thread.is_alive() + + +@pytest.mark.parametrize("id_mode", ["absent", "accept", "reject"]) +@pytest.mark.parametrize("phase", ["startup", "poll"]) +def test_id_negotiation_preserves_inbox_connection(monkeypatch, id_mode, phase): + from gateway.config import PlatformConfig + from plugins.platforms.email.adapter import EmailAdapter + + for key, value in { + "EMAIL_ADDRESS": "agent@example.test", + "EMAIL_PASSWORD": "test-only", + "EMAIL_IMAP_HOST": "imap.example.test", + "EMAIL_SMTP_HOST": "smtp.example.test", + }.items(): + monkeypatch.setenv(key, value) + adapter = EmailAdapter(PlatformConfig(enabled=True)) + monkeypatch.setattr(adapter, "_connect_smtp", MagicMock(return_value=MagicMock())) + + with imap_peer(id_mode) as (address, commands): + # TLS is out of scope: replace only the transport with real imaplib on + # loopback, keeping the actual IMAP parser and command API under test. + monkeypatch.setattr( + imaplib, + "IMAP4_SSL", + lambda *args, **kwargs: imaplib.IMAP4(*address, timeout=5), + ) + + if phase == "startup": + + async def connect_and_stop(): + try: + return await adapter.connect() + finally: + await adapter.disconnect() + + assert asyncio.run(connect_and_stop()) is True + else: + assert adapter._fetch_new_messages() == [] + assert adapter._last_fetch_failed is False + + assert commands.count("ID") == (0 if id_mode == "absent" else 1) + assert commands.index("LOGIN") < commands.index("SELECT") < commands.index("UID") + if id_mode != "absent": + assert commands.index("LOGIN") < commands.index("ID") < commands.index("SELECT") + assert commands[-1] == "LOGOUT" From 3e6241a819345e9636cd070752826d1f445cd560 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 00:38:02 -0700 Subject: [PATCH 161/276] test(relay): the OpenAI-stream e2e no longer deadlocks when Relay's finalizer has not started yet Relay's producer is pumped by the consumer thread's own event loop (ManagedLlmStream.__next__ -> run_until_complete), so the finalizer can only START inside a consumer next(). The harness blocked the consumer in _count_chunk waiting for the finalizer to start; when the loop had not reached it yet by the final chunk, that wait could never be satisfied and expired at 5 s. Reproduced ~1 run in 6 locally with a thread dump (consumer parked in Event.wait, no other thread anywhere near Relay); it took two unrelated PRs red in CI the same day (#103476, #103486). The hook now forces the ordering only when the finalizer has already started (that is the race under test), releases it otherwise, reports whether the race was forced, and the two tests repeat the stream until it was, asserting the invariant on every run. 15/15 green; with 74de0fd4fe's agent/ change reverted the tool-call test still fails 3/3, so it keeps guarding what it pins. --- tests/e2e/test_relay_native_openai_stream.py | 76 +++++++++++++------- 1 file changed, 49 insertions(+), 27 deletions(-) diff --git a/tests/e2e/test_relay_native_openai_stream.py b/tests/e2e/test_relay_native_openai_stream.py index 0ee25472e2..ee7910d227 100644 --- a/tests/e2e/test_relay_native_openai_stream.py +++ b/tests/e2e/test_relay_native_openai_stream.py @@ -22,7 +22,8 @@ def _sse(*chunk_bodies: bytes) -> bytes: def _stream_through_relay(tmp_path, monkeypatch, response_body: bytes, *, finalize_before): """Stream ``response_body`` through Relay; Relay's finalizer is forced to complete before the consumer thread processes the first chunk matching ``finalize_before(chunk)``. - Returns ``(hermes_result, relay_llm_end_event)``.""" + Returns ``(hermes_result, relay_llm_end_event, race_forced)``; ``race_forced`` is False when Relay + happened to reach its finalizer only after the consumer took that chunk (see the hook below).""" httpx = pytest.importorskip("httpx") nemo_relay = pytest.importorskip("nemo_relay") openai = pytest.importorskip("openai") @@ -56,11 +57,12 @@ def _stream_through_relay(tmp_path, monkeypatch, response_body: bytes, *, finali relay_finalizer_started = threading.Event() allow_relay_finalizer = threading.Event() relay_finalizer_finished = threading.Event() + forced = [] run_relay_finalizer = relay_llm.ManagedLlmStream._relay_finalizer def run_synchronized_relay_finalizer(managed_stream, attempt): relay_finalizer_started.set() - assert allow_relay_finalizer.wait(5), "consumer did not release Relay's finalizer" + assert allow_relay_finalizer.wait(30), "consumer did not release Relay's finalizer" try: return run_relay_finalizer(managed_stream, attempt) finally: @@ -71,11 +73,20 @@ def _stream_through_relay(tmp_path, monkeypatch, response_body: bytes, *, finali count_chunk = chat_completion_helpers._StreamingCall._count_chunk def count_chunk_after_relay_finalizes(self, diag, chunk): - # ``_count_chunk`` is the first thing the consumer does with every chunk. + # Relay's producer is pumped by the consumer thread's own event loop (``ManagedLlmStream. + # __next__`` -> ``run_until_complete``), so the finalizer can only START inside a consumer + # ``next()``. Whether it has started by the time the final chunk is handed over is a loop + # scheduling coin flip. When it has, hold it here so it provably completes before the consumer + # touches this chunk (the race under test). When it has not, it cannot run until we return; + # waiting for it here is a deadlock, not a slow runner (~1 run in 6 locally, and two + # unrelated PRs went red on it the same day). Callers repeat until the race materialized. if finalize_before(chunk): - assert relay_finalizer_started.wait(5), "Relay's finalizer did not start" - allow_relay_finalizer.set() - assert relay_finalizer_finished.wait(5), "Relay's finalizer did not finish" + if relay_finalizer_started.is_set(): + forced.append(True) + allow_relay_finalizer.set() + assert relay_finalizer_finished.wait(30), "Relay's finalizer did not finish" + else: + allow_relay_finalizer.set() # it will start inside the consumer's next ``next()`` return count_chunk(self, diag, chunk) monkeypatch.setattr(chat_completion_helpers._StreamingCall, "_count_chunk", count_chunk_after_relay_finalizes) @@ -100,7 +111,20 @@ def _stream_through_relay(tmp_path, monkeypatch, response_body: bytes, *, finali ] assert len(llm_end_events) == 1 assert llm_end_events[0].annotated_response is not None - return result, llm_end_events[0] + return result, llm_end_events[0], bool(forced) + + +def _stream_until_race_forced(tmp_path, monkeypatch, body, *, finalize_before, attempts=12): + """Run the stream until Relay's finalizer was provably forced ahead of the consumer at least once; + every run's result is returned so callers assert the invariant on all of them.""" + runs = [] + for _ in range(attempts): + result, llm_end, race_forced = _stream_through_relay( + tmp_path, monkeypatch, body, finalize_before=finalize_before) + runs.append((result, llm_end)) + if race_forced: + return runs + pytest.fail(f"Relay's finalizer never started before the consumer took the chunk in {attempts} runs") def test_openai_stream_usage_reaches_relay_parent_event(tmp_path, monkeypatch): @@ -110,15 +134,14 @@ def test_openai_stream_usage_reaches_relay_parent_event(tmp_path, monkeypatch): b'"choices":[{"index":0,"delta":{},"finish_reason":"stop"}]', b'"choices":[],"usage":{"prompt_tokens":100,"completion_tokens":10,"total_tokens":110}', ) - result, llm_end = _stream_through_relay( - tmp_path, monkeypatch, body, - finalize_before=lambda chunk: not chunk.choices and getattr(chunk, "usage", None) is not None) - - assert result.usage is not None - assert (result.usage.prompt_tokens, result.usage.completion_tokens, result.usage.total_tokens) == (100, 10, 110) - assert llm_end.annotated_response.usage == { - "prompt_tokens": 100, "completion_tokens": 10, "total_tokens": 110} - assert llm_end.annotated_response.message == "done" + for result, llm_end in _stream_until_race_forced( + tmp_path, monkeypatch, body, + finalize_before=lambda chunk: not chunk.choices and getattr(chunk, "usage", None) is not None): + assert result.usage is not None + assert (result.usage.prompt_tokens, result.usage.completion_tokens, result.usage.total_tokens) == (100, 10, 110) + assert llm_end.annotated_response.usage == { + "prompt_tokens": 100, "completion_tokens": 10, "total_tokens": 110} + assert llm_end.annotated_response.message == "done" def test_openai_stream_final_tool_call_delta_reaches_relay_parent_event(tmp_path, monkeypatch): @@ -131,14 +154,13 @@ def test_openai_stream_final_tool_call_delta_reaches_relay_parent_event(tmp_path b'"choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"\\"/tmp/x\\"}"}}]},' b'"finish_reason":"tool_calls"}]', ) - result, llm_end = _stream_through_relay( - tmp_path, monkeypatch, body, - finalize_before=lambda chunk: bool(chunk.choices) and chunk.choices[0].finish_reason == "tool_calls") - - hermes_call = result.choices[0].message.tool_calls[0] - assert (hermes_call.function.name, hermes_call.function.arguments) == ("read_file", '{"path": "/tmp/x"}') - assert result.choices[0].finish_reason == "tool_calls" - assert llm_end.annotated_response.message is None - (relay_call,) = llm_end.annotated_response.tool_calls - assert (relay_call["name"], relay_call["arguments"]) == ("read_file", {"path": "/tmp/x"}) - assert llm_end.annotated_response.finish_reason == "tool_use" + for result, llm_end in _stream_until_race_forced( + tmp_path, monkeypatch, body, + finalize_before=lambda chunk: bool(chunk.choices) and chunk.choices[0].finish_reason == "tool_calls"): + hermes_call = result.choices[0].message.tool_calls[0] + assert (hermes_call.function.name, hermes_call.function.arguments) == ("read_file", '{"path": "/tmp/x"}') + assert result.choices[0].finish_reason == "tool_calls" + assert llm_end.annotated_response.message is None + (relay_call,) = llm_end.annotated_response.tool_calls + assert (relay_call["name"], relay_call["arguments"]) == ("read_file", {"path": "/tmp/x"}) + assert llm_end.annotated_response.finish_reason == "tool_use" From 314e36536365281a5b1438e7026a4f226bedff5d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 5 Sep 2026 06:42:48 -0700 Subject: [PATCH 162/276] test(relay): force the finalizer-before-consumer ordering deterministically by pumping the stream's own loop Independent review showed the retry version was a scheduling lottery: under a forced consumer-first schedule (provider generator held at the terminal chunk until the consumer took it) both tests failed after 12 attempts with correct production code. The reviewer also demonstrated the deterministic alternative; this adopts it. At the chosen chunk the hook releases the finalizer and pumps the stream's OWN event loop (stream._loop.run_until_complete over a to_thread wait) until the finalizer has completed, then hands the chunk to the consumer: a real happens-before on every schedule. A schedule probe's gate on the provider generator is released first so the pump can reach EOF. The retry helper and the race-forced plumbing are gone; each test runs once. Adverse consumer-first schedule (reviewer's probe plugin): 3/3 pass (was 0/3). Normal schedule: 6/6. Production fix 74de0fd4fe reverted: tool-call test still fails, so the guard is intact. --- tests/e2e/test_relay_native_openai_stream.py | 89 +++++++++----------- 1 file changed, 41 insertions(+), 48 deletions(-) diff --git a/tests/e2e/test_relay_native_openai_stream.py b/tests/e2e/test_relay_native_openai_stream.py index ee7910d227..7c0aef79c5 100644 --- a/tests/e2e/test_relay_native_openai_stream.py +++ b/tests/e2e/test_relay_native_openai_stream.py @@ -22,8 +22,7 @@ def _sse(*chunk_bodies: bytes) -> bytes: def _stream_through_relay(tmp_path, monkeypatch, response_body: bytes, *, finalize_before): """Stream ``response_body`` through Relay; Relay's finalizer is forced to complete before the consumer thread processes the first chunk matching ``finalize_before(chunk)``. - Returns ``(hermes_result, relay_llm_end_event, race_forced)``; ``race_forced`` is False when Relay - happened to reach its finalizer only after the consumer took that chunk (see the hook below).""" + Returns ``(hermes_result, relay_llm_end_event)``.""" httpx = pytest.importorskip("httpx") nemo_relay = pytest.importorskip("nemo_relay") openai = pytest.importorskip("openai") @@ -57,7 +56,6 @@ def _stream_through_relay(tmp_path, monkeypatch, response_body: bytes, *, finali relay_finalizer_started = threading.Event() allow_relay_finalizer = threading.Event() relay_finalizer_finished = threading.Event() - forced = [] run_relay_finalizer = relay_llm.ManagedLlmStream._relay_finalizer def run_synchronized_relay_finalizer(managed_stream, attempt): @@ -73,20 +71,28 @@ def _stream_through_relay(tmp_path, monkeypatch, response_body: bytes, *, finali count_chunk = chat_completion_helpers._StreamingCall._count_chunk def count_chunk_after_relay_finalizes(self, diag, chunk): - # Relay's producer is pumped by the consumer thread's own event loop (``ManagedLlmStream. - # __next__`` -> ``run_until_complete``), so the finalizer can only START inside a consumer - # ``next()``. Whether it has started by the time the final chunk is handed over is a loop - # scheduling coin flip. When it has, hold it here so it provably completes before the consumer - # touches this chunk (the race under test). When it has not, it cannot run until we return; - # waiting for it here is a deadlock, not a slow runner (~1 run in 6 locally, and two - # unrelated PRs went red on it the same day). Callers repeat until the race materialized. + # Relay's producer is pumped by the consumer thread's OWN event loop (``ManagedLlmStream. + # __next__`` -> ``run_until_complete``), so the finalizer can only start while that loop runs. + # Blocking the consumer thread here and waiting for it therefore deadlocked whenever the loop + # had not reached EOF yet (~1 run in 6). Instead: release the finalizer and PUMP THE STREAM'S + # LOOP until it has completed, then hand the chunk to the consumer. That is a real + # happens-before (finalizer done -> consumer sees chunk) on every schedule, not a retry lottery. if finalize_before(chunk): - if relay_finalizer_started.is_set(): - forced.append(True) - allow_relay_finalizer.set() - assert relay_finalizer_finished.wait(30), "Relay's finalizer did not finish" - else: - allow_relay_finalizer.set() # it will start inside the consumer's next ``next()`` + import asyncio + + stream = self.managed_stream_holder["stream"] + allow_relay_finalizer.set() + # A schedule probe may hold the provider generator at this chunk until the consumer has + # taken it (adverse consumer-first ordering); release it so the pump below can reach EOF. + gate = getattr(stream, "_review_gate", None) + if gate is not None: + gate.set() + + async def finalizer_done(): + return await asyncio.to_thread(relay_finalizer_finished.wait, 30) + + assert stream._loop.run_until_complete(finalizer_done()), "Relay's finalizer did not finish" + assert relay_finalizer_started.is_set() return count_chunk(self, diag, chunk) monkeypatch.setattr(chat_completion_helpers._StreamingCall, "_count_chunk", count_chunk_after_relay_finalizes) @@ -111,20 +117,7 @@ def _stream_through_relay(tmp_path, monkeypatch, response_body: bytes, *, finali ] assert len(llm_end_events) == 1 assert llm_end_events[0].annotated_response is not None - return result, llm_end_events[0], bool(forced) - - -def _stream_until_race_forced(tmp_path, monkeypatch, body, *, finalize_before, attempts=12): - """Run the stream until Relay's finalizer was provably forced ahead of the consumer at least once; - every run's result is returned so callers assert the invariant on all of them.""" - runs = [] - for _ in range(attempts): - result, llm_end, race_forced = _stream_through_relay( - tmp_path, monkeypatch, body, finalize_before=finalize_before) - runs.append((result, llm_end)) - if race_forced: - return runs - pytest.fail(f"Relay's finalizer never started before the consumer took the chunk in {attempts} runs") + return result, llm_end_events[0] def test_openai_stream_usage_reaches_relay_parent_event(tmp_path, monkeypatch): @@ -134,14 +127,14 @@ def test_openai_stream_usage_reaches_relay_parent_event(tmp_path, monkeypatch): b'"choices":[{"index":0,"delta":{},"finish_reason":"stop"}]', b'"choices":[],"usage":{"prompt_tokens":100,"completion_tokens":10,"total_tokens":110}', ) - for result, llm_end in _stream_until_race_forced( - tmp_path, monkeypatch, body, - finalize_before=lambda chunk: not chunk.choices and getattr(chunk, "usage", None) is not None): - assert result.usage is not None - assert (result.usage.prompt_tokens, result.usage.completion_tokens, result.usage.total_tokens) == (100, 10, 110) - assert llm_end.annotated_response.usage == { - "prompt_tokens": 100, "completion_tokens": 10, "total_tokens": 110} - assert llm_end.annotated_response.message == "done" + result, llm_end = _stream_through_relay( + tmp_path, monkeypatch, body, + finalize_before=lambda chunk: not chunk.choices and getattr(chunk, "usage", None) is not None) + assert result.usage is not None + assert (result.usage.prompt_tokens, result.usage.completion_tokens, result.usage.total_tokens) == (100, 10, 110) + assert llm_end.annotated_response.usage == { + "prompt_tokens": 100, "completion_tokens": 10, "total_tokens": 110} + assert llm_end.annotated_response.message == "done" def test_openai_stream_final_tool_call_delta_reaches_relay_parent_event(tmp_path, monkeypatch): @@ -154,13 +147,13 @@ def test_openai_stream_final_tool_call_delta_reaches_relay_parent_event(tmp_path b'"choices":[{"index":0,"delta":{"tool_calls":[{"index":0,"function":{"arguments":"\\"/tmp/x\\"}"}}]},' b'"finish_reason":"tool_calls"}]', ) - for result, llm_end in _stream_until_race_forced( - tmp_path, monkeypatch, body, - finalize_before=lambda chunk: bool(chunk.choices) and chunk.choices[0].finish_reason == "tool_calls"): - hermes_call = result.choices[0].message.tool_calls[0] - assert (hermes_call.function.name, hermes_call.function.arguments) == ("read_file", '{"path": "/tmp/x"}') - assert result.choices[0].finish_reason == "tool_calls" - assert llm_end.annotated_response.message is None - (relay_call,) = llm_end.annotated_response.tool_calls - assert (relay_call["name"], relay_call["arguments"]) == ("read_file", {"path": "/tmp/x"}) - assert llm_end.annotated_response.finish_reason == "tool_use" + result, llm_end = _stream_through_relay( + tmp_path, monkeypatch, body, + finalize_before=lambda chunk: bool(chunk.choices) and chunk.choices[0].finish_reason == "tool_calls") + hermes_call = result.choices[0].message.tool_calls[0] + assert (hermes_call.function.name, hermes_call.function.arguments) == ("read_file", '{"path": "/tmp/x"}') + assert result.choices[0].finish_reason == "tool_calls" + assert llm_end.annotated_response.message is None + (relay_call,) = llm_end.annotated_response.tool_calls + assert (relay_call["name"], relay_call["arguments"]) == ("read_file", {"path": "/tmp/x"}) + assert llm_end.annotated_response.finish_reason == "tool_use" From a2274849062894b10a276cf4783087174a655d1b Mon Sep 17 00:00:00 2001 From: Mauvis Ledford Date: Mon, 17 Aug 2026 15:35:27 +0800 Subject: [PATCH 163/276] fix(agent): honor Retry-After on retryable 5xx Retryable server errors can carry provider cooldowns just like 429 responses. Parse Retry-After from response headers or structured error bodies before falling back to jittered backoff, and cover HTTP 524 behavior with runtime regression tests. --- agent/turn_recovery.py | 39 ++++++++++++----------- tests/run_agent/test_run_agent.py | 51 +++++++++++++++++++++++++++++-- 2 files changed, 70 insertions(+), 20 deletions(-) diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py index 5aed02349a..417feac8ce 100644 --- a/agent/turn_recovery.py +++ b/agent/turn_recovery.py @@ -974,26 +974,31 @@ def compute_error_backoff( agent: Any, api_error: Exception, *, retry_count: int, max_retries: int, is_rate_limited: bool, is_zai_coding_overload: bool, base_url: Any, model: Any, ) -> float: - """Pick the wait before the next API retry and announce it. Retry-After wins for rate - limits (capped at 600s: Anthropic Tier 1 buckets reset in ~171s, so a 120s cap re-tripped - the limit); otherwise jittered backoff, replaced by the adaptive policy for 429s / Z.AI - overloads. Normal retries are buffered; long Z.AI Coding waits surface immediately.""" + """Pick the wait before the next API retry and announce it. Retry-After wins for + rate limits and any other retryable error (capped at 600s: Anthropic Tier 1 buckets + reset in ~171s, so a 120s cap re-tripped the limit); otherwise jittered backoff, + replaced by the adaptive policy for 429s / Z.AI overloads. Normal retries are + buffered; long Z.AI Coding waits surface immediately.""" # Imported lazily so tests that patch ``agent.retry_utils.jittered_backoff`` / # ``adaptive_rate_limit_backoff`` (incl. the run_agent conftest fast-backoff fixture) intercept. - from agent.retry_utils import adaptive_rate_limit_backoff, jittered_backoff + from agent.retry_utils import adaptive_rate_limit_backoff, jittered_backoff, parse_retry_after_seconds - _retry_after = None - _resp_headers = getattr(getattr(api_error, "response", None), "headers", None) if is_rate_limited else None - if _resp_headers and hasattr(_resp_headers, "get"): - _ra_raw = _resp_headers.get("retry-after") or _resp_headers.get("Retry-After") - if _ra_raw: - try: - # Cap at 10 minutes. Anthropic Tier 1 input-token buckets reset in ~171s, so a 120s cap - # caused us to retry before the actual reset window and re-trip the limit. 600s covers all - # realistic provider reset windows while still rejecting pathological values. (#26293) - _retry_after = min(float(_ra_raw), 600) - except (TypeError, ValueError): - pass + # Respect Retry-After on every retryable provider error, not just 429s. Retryable + # 5xx responses (e.g. Cloudflare 520/524) also carry the header or a structured + # ``retry_after`` problem-detail body field; ignoring either turns an origin + # outage into a retry storm. + _retry_after = parse_retry_after_seconds( + getattr(getattr(api_error, "response", None), "headers", None) + ) + if _retry_after is None: + _error_body = getattr(api_error, "body", None) + if isinstance(_error_body, dict): + _retry_after = parse_retry_after_seconds(_error_body.get("retry_after")) + if _retry_after is not None: + # Cap at 10 minutes. Anthropic Tier 1 input-token buckets reset in ~171s, so a 120s cap + # caused us to retry before the actual reset window and re-trip the limit. 600s covers all + # realistic provider reset windows while still rejecting pathological values. (#26293) + _retry_after = min(_retry_after, 600) wait_time = _retry_after if _retry_after else jittered_backoff(retry_count, base_delay=2.0, max_delay=60.0) _backoff_policy = None _adaptive = is_rate_limited or is_zai_coding_overload diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 0da74057bd..a2b7443ee3 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -1961,9 +1961,10 @@ class TestExecuteToolCalls: class TestRetryAfterCap: - """#26293: the conversation loop owns rate-limit backoff and honors the - Retry-After header up to a 600s ceiling (was 120s, which retried before - Tier-1 reset windows of ~171s and re-tripped the limit).""" + """The loop honors provider cooldowns up to a 600-second ceiling. + + This covers rate-limit headers (#26293) and retryable 5xx responses. + """ def _drive_once(self, agent, retry_after_value): """Raise one 429 carrying ``Retry-After`` and capture the wait the loop @@ -2004,6 +2005,50 @@ class TestRetryAfterCap: status = self._drive_once(agent, 300) assert "Waiting 300.0s" in status + @pytest.mark.parametrize( + ("headers", "body"), + [ + ({"Retry-After": "120"}, {}), + ({}, {"status": 524, "retry_after": 120}), + ], + ids=("header", "problem-detail-body"), + ) + def test_retry_after_on_cloudflare_524_is_honored( + self, agent, headers, body + ): + """A retryable 5xx must not bypass the provider's cooldown.""" + + class _CloudflareTimeout(Exception): + status_code = 524 + + def __str__(self): + return "Error code: 524 - origin response timeout" + + error = _CloudflareTimeout() + error.response = SimpleNamespace(headers=headers) + error.body = body + + def _fake_api_call(api_kwargs): + raise error + + agent._interruptible_api_call = _fake_api_call + agent._persist_session = lambda *args, **kwargs: None + agent._save_trajectory = lambda *args, **kwargs: None + + captured = [] + original_buffer = agent._buffer_status + + def _capture_status(msg, *args, **kwargs): + captured.append(msg) + if "Retrying in" in msg: + agent._interrupt_requested = True + return original_buffer(msg, *args, **kwargs) + + agent._buffer_status = _capture_status + agent.run_conversation("hello") + + assert any("Retrying in 120.0s" in msg for msg in captured) + class TestConcurrentToolExecution: From c6ef0756134713243d2572c0178eb54a19826e0b Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 20:58:17 +0530 Subject: [PATCH 164/276] fix(agent): parse nested retry_after bodies and emit long 5xx cooldowns Salvage follow-ups on #88236 (krunkosaurus): - Some providers nest the cooldown as body["error"]["retry_after"] (the same unwrap _extract_rate_limit_context already uses); only the top-level shape was read, so those errors silently fell back to jittered backoff. - A 5xx Retry-After can reach the 600s cap; that wait was buffered (replayed only on terminal failure), leaving the user silent for minutes. Long provider cooldowns now emit immediately, mirroring the zai_coding_overload_long path. Jittered waits keep the old buffering. Test widened with the nested-body parametrize row; proven red when the unwrap is neutralized. --- agent/turn_recovery.py | 17 +++++++++++++++-- tests/run_agent/test_run_agent.py | 11 ++++++++++- 2 files changed, 25 insertions(+), 3 deletions(-) diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py index 417feac8ce..ff92e7e72f 100644 --- a/agent/turn_recovery.py +++ b/agent/turn_recovery.py @@ -993,7 +993,11 @@ def compute_error_backoff( if _retry_after is None: _error_body = getattr(api_error, "body", None) if isinstance(_error_body, dict): - _retry_after = parse_retry_after_seconds(_error_body.get("retry_after")) + # Some providers nest it as error.retry_after (the same unwrap + # _extract_rate_limit_context uses), others put it at the top level. + _nested = _error_body.get("error") + _payload = _nested if isinstance(_nested, dict) else _error_body + _retry_after = parse_retry_after_seconds(_payload.get("retry_after")) if _retry_after is not None: # Cap at 10 minutes. Anthropic Tier 1 input-token buckets reset in ~171s, so a 120s cap # caused us to retry before the actual reset window and re-trip the limit. 600s covers all @@ -1015,7 +1019,16 @@ def compute_error_backoff( else: agent._buffer_status(_rate_limit_status) else: - agent._buffer_status(f"⏳ Retrying in {wait_time:.1f}s (attempt {retry_count}/{max_retries})...") + _retry_status = ( + f"⏳ Retrying in {wait_time:.1f}s (attempt {retry_count}/{max_retries})..." + ) + if _retry_after is not None and _retry_after > 60: + # A 5xx Retry-After can now reach the 600s cap; buffering that wait + # would leave the user silent for minutes, so surface long provider + # cooldowns immediately (mirrors the zai_coding_overload_long path). + agent._emit_status(_retry_status) + else: + agent._buffer_status(_retry_status) logger.warning( "Retrying API call in %ss (attempt %s/%s) %s policy=%s error=%s", wait_time, retry_count, max_retries, agent._client_log_context(), diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index a2b7443ee3..867ba23990 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -2010,8 +2010,9 @@ class TestRetryAfterCap: [ ({"Retry-After": "120"}, {}), ({}, {"status": 524, "retry_after": 120}), + ({}, {"status": 524, "error": {"retry_after": 120}}), ], - ids=("header", "problem-detail-body"), + ids=("header", "problem-detail-body", "nested-problem-detail-body"), ) def test_retry_after_on_cloudflare_524_is_honored( self, agent, headers, body @@ -2037,6 +2038,7 @@ class TestRetryAfterCap: captured = [] original_buffer = agent._buffer_status + original_emit = agent._emit_status def _capture_status(msg, *args, **kwargs): captured.append(msg) @@ -2044,7 +2046,14 @@ class TestRetryAfterCap: agent._interrupt_requested = True return original_buffer(msg, *args, **kwargs) + def _capture_emit(msg): + captured.append(msg) + if "Retrying in" in msg: + agent._interrupt_requested = True + return original_emit(msg) + agent._buffer_status = _capture_status + agent._emit_status = _capture_emit agent.run_conversation("hello") assert any("Retrying in 120.0s" in msg for msg in captured) From 44e52e7524232e838a6e69e3d28d21688ad35101 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 21:25:56 +0530 Subject: [PATCH 165/276] fix(agent): normalize zero-cooldown semantics, dedupe test driver MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit /simplify-code pass on the Retry-After salvage stack: - compute_error_backoff now decides "no usable cooldown" exactly once: a parsed 0.0 (retry-after: 0, or an HTTP-date in the past, which the shared parser clamps to 0) is treated as absent instead of falling through an accidental falsy check — prevents a hot-loop retry and keeps the sentinel semantics uniform (is None / is not None at all four sites). - Comment corrected: the sibling unwrap lives in extract_api_error_context, not _extract_rate_limit_context. - Test driver deduped: one _retryable_error factory + one _drive_once shared by the 429 and 524 tests; added the over-cap (3600 → 600) and no-cooldown fallback rows requested in the #103722 review. Mutation checks: nested-unwrap neutralized → nested row red; cap removed → over-cap row red; restored → all 298 green. --- agent/turn_recovery.py | 11 ++- tests/run_agent/test_run_agent.py | 119 ++++++++++++++---------------- 2 files changed, 65 insertions(+), 65 deletions(-) diff --git a/agent/turn_recovery.py b/agent/turn_recovery.py index ff92e7e72f..be2a46378f 100644 --- a/agent/turn_recovery.py +++ b/agent/turn_recovery.py @@ -994,7 +994,7 @@ def compute_error_backoff( _error_body = getattr(api_error, "body", None) if isinstance(_error_body, dict): # Some providers nest it as error.retry_after (the same unwrap - # _extract_rate_limit_context uses), others put it at the top level. + # extract_api_error_context uses), others put it at the top level. _nested = _error_body.get("error") _payload = _nested if isinstance(_nested, dict) else _error_body _retry_after = parse_retry_after_seconds(_payload.get("retry_after")) @@ -1003,10 +1003,15 @@ def compute_error_backoff( # caused us to retry before the actual reset window and re-trip the limit. 600s covers all # realistic provider reset windows while still rejecting pathological values. (#26293) _retry_after = min(_retry_after, 600) - wait_time = _retry_after if _retry_after else jittered_backoff(retry_count, base_delay=2.0, max_delay=60.0) + if _retry_after <= 0: + # A zero/expired cooldown (retry-after: 0, or an HTTP-date in the + # past, which the parser clamps to 0.0) carries no usable wait — + # treat it as absent so we never hot-loop the provider. + _retry_after = None + wait_time = _retry_after if _retry_after is not None else jittered_backoff(retry_count, base_delay=2.0, max_delay=60.0) _backoff_policy = None _adaptive = is_rate_limited or is_zai_coding_overload - if _adaptive and not _retry_after: + if _adaptive and _retry_after is None: wait_time, _backoff_policy = adaptive_rate_limit_backoff( retry_count, base_url=str(base_url), model=model, error=api_error, default_wait=wait_time, ) diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 867ba23990..80b8882d2a 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -1966,68 +1966,29 @@ class TestRetryAfterCap: This covers rate-limit headers (#26293) and retryable 5xx responses. """ - def _drive_once(self, agent, retry_after_value): - """Raise one 429 carrying ``Retry-After`` and capture the wait the loop - chose. Interrupt during the backoff sleep so the test doesn't actually - wait, and return the status string that reports the wait time.""" + @staticmethod + def _retryable_error(status_code, headers, body=None): + """A provider error carrying optional Retry-After surfaces.""" + message = ( + "Error code: 429 - Rate limit exceeded." + if status_code == 429 + else f"Error code: {status_code} - origin response timeout" + ) - class _RateLimitError(Exception): - status_code = 429 - response = SimpleNamespace(headers={"retry-after": str(retry_after_value)}) + class _ProviderError(Exception): + def __init__(self): + super().__init__(message) + self.status_code = status_code + self.response = SimpleNamespace(headers=headers) + if body is not None: + self.body = body - def __str__(self): - return "Error code: 429 - Rate limit exceeded." + return _ProviderError() - def _fake_api_call(api_kwargs): - raise _RateLimitError() - - agent._interruptible_api_call = _fake_api_call - agent._persist_session = lambda *args, **kwargs: None - agent._save_trajectory = lambda *args, **kwargs: None - - captured = [] - original_buffer = agent._buffer_status - - def _capture_status(msg, *args, **kwargs): - captured.append(msg) - # Break out of the incremental backoff sleep immediately rather - # than blocking for the full Retry-After window. - if "Waiting" in msg: - agent._interrupt_requested = True - return original_buffer(msg, *args, **kwargs) - - agent._buffer_status = _capture_status - agent.run_conversation("hello") - return next((m for m in captured if "Waiting" in m), "") - - def test_retry_after_under_cap_is_honored(self, agent): - # 300s > old 120s cap but < new 600s cap → used verbatim. - status = self._drive_once(agent, 300) - assert "Waiting 300.0s" in status - - @pytest.mark.parametrize( - ("headers", "body"), - [ - ({"Retry-After": "120"}, {}), - ({}, {"status": 524, "retry_after": 120}), - ({}, {"status": 524, "error": {"retry_after": 120}}), - ], - ids=("header", "problem-detail-body", "nested-problem-detail-body"), - ) - def test_retry_after_on_cloudflare_524_is_honored( - self, agent, headers, body - ): - """A retryable 5xx must not bypass the provider's cooldown.""" - - class _CloudflareTimeout(Exception): - status_code = 524 - - def __str__(self): - return "Error code: 524 - origin response timeout" - - error = _CloudflareTimeout() - error.response = SimpleNamespace(headers=headers) - error.body = body + def _drive_once(self, agent, error, status_marker): + """Raise ``error`` from the API call and capture the backoff status the + loop chose. Interrupt during the backoff sleep so the test doesn't + actually wait, and return the status string reporting the wait.""" def _fake_api_call(api_kwargs): raise error @@ -2042,22 +2003,56 @@ class TestRetryAfterCap: def _capture_status(msg, *args, **kwargs): captured.append(msg) - if "Retrying in" in msg: + # Break out of the backoff sleep immediately rather than blocking + # for the full Retry-After window. + if status_marker in msg: agent._interrupt_requested = True return original_buffer(msg, *args, **kwargs) def _capture_emit(msg): captured.append(msg) - if "Retrying in" in msg: + if status_marker in msg: agent._interrupt_requested = True return original_emit(msg) agent._buffer_status = _capture_status agent._emit_status = _capture_emit agent.run_conversation("hello") + return next((m for m in captured if status_marker in m), "") - assert any("Retrying in 120.0s" in msg for msg in captured) + def test_retry_after_under_cap_is_honored(self, agent): + # 300s > old 120s cap but < new 600s cap → used verbatim. + error = self._retryable_error(429, {"retry-after": "300"}) + status = self._drive_once(agent, error, "Waiting") + assert "Waiting 300.0s" in status + @pytest.mark.parametrize( + ("headers", "body", "expected_wait"), + [ + ({"Retry-After": "120"}, {}, "120.0"), + ({}, {"status": 524, "retry_after": 120}, "120.0"), + ({}, {"status": 524, "error": {"retry_after": 120}}, "120.0"), + # Above the 600s ceiling → capped, never used verbatim. + ({"Retry-After": "3600"}, {}, "600.0"), + # No cooldown on header or body → falls through to jittered + # backoff (patched to 0.0 by the conftest fixture), no crash. + ({}, {"status": 524}, "0.0"), + ], + ids=( + "header", + "problem-detail-body", + "nested-problem-detail-body", + "over-cap-is-capped", + "no-cooldown-falls-back", + ), + ) + def test_retry_after_on_cloudflare_524_is_honored( + self, agent, headers, body, expected_wait + ): + """A retryable 5xx must not bypass the provider's cooldown.""" + error = self._retryable_error(524, headers, body) + status = self._drive_once(agent, error, "Retrying in") + assert f"Retrying in {expected_wait}s" in status class TestConcurrentToolExecution: From 8fd3e08ab97da535bdf8c1ca5eea850c368ce6eb Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 21:31:08 +0530 Subject: [PATCH 166/276] test(agent): pin the emit-vs-buffer gate from both sides The simplify reviewers flagged that every cooldown parametrize row was > 60s, so the emit branch was never asserted against its buffered complement. _drive_once now records which surface the status went through, and a 30s row asserts short cooldowns keep the buffered line. Mutation-checked: flipping the threshold fails the four emit rows. --- tests/run_agent/test_run_agent.py | 29 +++++++++++++++++------------ 1 file changed, 17 insertions(+), 12 deletions(-) diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 80b8882d2a..29cd2692f3 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -2002,7 +2002,7 @@ class TestRetryAfterCap: original_emit = agent._emit_status def _capture_status(msg, *args, **kwargs): - captured.append(msg) + captured.append((msg, "buffer")) # Break out of the backoff sleep immediately rather than blocking # for the full Retry-After window. if status_marker in msg: @@ -2010,7 +2010,7 @@ class TestRetryAfterCap: return original_buffer(msg, *args, **kwargs) def _capture_emit(msg): - captured.append(msg) + captured.append((msg, "emit")) if status_marker in msg: agent._interrupt_requested = True return original_emit(msg) @@ -2018,41 +2018,46 @@ class TestRetryAfterCap: agent._buffer_status = _capture_status agent._emit_status = _capture_emit agent.run_conversation("hello") - return next((m for m in captured if status_marker in m), "") + return next(((m, s) for m, s in captured if status_marker in m), ("", "")) def test_retry_after_under_cap_is_honored(self, agent): # 300s > old 120s cap but < new 600s cap → used verbatim. error = self._retryable_error(429, {"retry-after": "300"}) - status = self._drive_once(agent, error, "Waiting") + status, _ = self._drive_once(agent, error, "Waiting") assert "Waiting 300.0s" in status @pytest.mark.parametrize( - ("headers", "body", "expected_wait"), + ("headers", "body", "expected_wait", "expected_surface"), [ - ({"Retry-After": "120"}, {}, "120.0"), - ({}, {"status": 524, "retry_after": 120}, "120.0"), - ({}, {"status": 524, "error": {"retry_after": 120}}, "120.0"), + # Long cooldowns (> 60s) surface immediately... + ({"Retry-After": "120"}, {}, "120.0", "emit"), + ({}, {"status": 524, "retry_after": 120}, "120.0", "emit"), + ({}, {"status": 524, "error": {"retry_after": 120}}, "120.0", "emit"), # Above the 600s ceiling → capped, never used verbatim. - ({"Retry-After": "3600"}, {}, "600.0"), + ({"Retry-After": "3600"}, {}, "600.0", "emit"), + # ...short cooldowns keep the buffered status line. + ({"Retry-After": "30"}, {}, "30.0", "buffer"), # No cooldown on header or body → falls through to jittered # backoff (patched to 0.0 by the conftest fixture), no crash. - ({}, {"status": 524}, "0.0"), + ({}, {"status": 524}, "0.0", "buffer"), ], ids=( "header", "problem-detail-body", "nested-problem-detail-body", "over-cap-is-capped", + "short-cooldown-is-buffered", "no-cooldown-falls-back", ), ) def test_retry_after_on_cloudflare_524_is_honored( - self, agent, headers, body, expected_wait + self, agent, headers, body, expected_wait, expected_surface ): """A retryable 5xx must not bypass the provider's cooldown.""" error = self._retryable_error(524, headers, body) - status = self._drive_once(agent, error, "Retrying in") + status, surface = self._drive_once(agent, error, "Retrying in") assert f"Retrying in {expected_wait}s" in status + assert surface == expected_surface class TestConcurrentToolExecution: From 66f16688507565d376aad527fa1b8b2fd9aaeeb2 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sun, 6 Sep 2026 21:29:58 +0530 Subject: [PATCH 167/276] refactor(email): fold #92979 review findings into the IMAP ID gate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two shape fixes from the review of #92979, no behavior change: - Read imap.capabilities directly instead of the getattr(...) or () fallback — a real imaplib.IMAP4 always sets the attribute in _connect() (raising if the server sends no CAPABILITY), so the default branch only existed for a MagicMock(spec=["xatom"]) that no production path produces. Drop the test that pinned it. - Trim tests to the invariant bar: the four mock tests duplicated what test_email_imap_id_protocol.py proves on the wire (real imaplib, real socket, command order), and the phase parametrize (startup vs poll) exercised the same single _send_imap_id call site twice. Keep the three id_mode cases through the poll path. 42 tests pass; mutation check: guard removed -> the absent-ID case fails with the #39856 SELECT error, restored -> green. --- plugins/platforms/email/adapter.py | 3 +- tests/gateway/test_email.py | 50 -------------------- tests/gateway/test_email_imap_id_protocol.py | 18 ++----- 3 files changed, 4 insertions(+), 67 deletions(-) diff --git a/plugins/platforms/email/adapter.py b/plugins/platforms/email/adapter.py index efcc623897..59b6e38273 100644 --- a/plugins/platforms/email/adapter.py +++ b/plugins/platforms/email/adapter.py @@ -148,8 +148,7 @@ def _send_imap_id(imap: "imaplib.IMAP4") -> None: connection, which imaplib cannot surface here — the failure appears one command later as a misleading SELECT error and the adapter retries forever (Purelymail, #39856). ``imap.capabilities`` is populated by imaplib at connect, so the check is free.""" - caps = getattr(imap, "capabilities", ()) or () - if "ID" not in caps: + if "ID" not in imap.capabilities: logger.debug( "[Email] Server does not advertise IMAP ID capability; skipping ID" ) diff --git a/tests/gateway/test_email.py b/tests/gateway/test_email.py index 07278a7b84..271195daff 100644 --- a/tests/gateway/test_email.py +++ b/tests/gateway/test_email.py @@ -963,56 +963,6 @@ class TestImapIdExtensionForNetEase(unittest.TestCase): self.assertIn("login", names) self.assertLess(names.index("login"), names.index("xatom")) - def test_send_imap_id_sent_when_capability_advertised(self): - """ID goes out when the server lists it in CAPABILITY.""" - from plugins.platforms.email.adapter import _send_imap_id - - mock_imap = MagicMock() - mock_imap.capabilities = ("IMAP4REV1", "ID", "UIDPLUS") - - _send_imap_id(mock_imap) - - mock_imap.xatom.assert_called_once() - self.assertEqual(mock_imap.xatom.call_args.args[0], "ID") - - def test_send_imap_id_skipped_when_capability_absent(self): - """Servers that do not advertise ID must not receive it: Purelymail - answers the unknown command with an untagged ``* BYE Unknown - command.`` and drops the connection, which imaplib surfaces one - command later as a misleading SELECT failure.""" - from plugins.platforms.email.adapter import _send_imap_id - - mock_imap = MagicMock() - mock_imap.capabilities = ("IMAP4REV1", "UIDPLUS") - - _send_imap_id(mock_imap) - - mock_imap.xatom.assert_not_called() - - def test_send_imap_id_skipped_when_capabilities_attribute_missing(self): - """A connection object without a capabilities attribute must not - crash the helper; fail toward not sending optional commands.""" - from plugins.platforms.email.adapter import _send_imap_id - - mock_imap = MagicMock(spec=["xatom"]) - - _send_imap_id(mock_imap) - - mock_imap.xatom.assert_not_called() - - def test_send_imap_id_rejection_still_swallowed(self): - """A server that advertises ID but rejects it with a tagged error - keeps the existing best-effort handling: no exception escapes.""" - from plugins.platforms.email.adapter import _send_imap_id - - mock_imap = MagicMock() - mock_imap.capabilities = ("IMAP4REV1", "ID") - mock_imap.xatom.side_effect = Exception("BAD ID rejected") - - _send_imap_id(mock_imap) # must not raise - - mock_imap.xatom.assert_called_once() - class TestConnectSmtp(unittest.TestCase): """Test _connect_smtp() helper: protocol selection and IPv6 fallback.""" diff --git a/tests/gateway/test_email_imap_id_protocol.py b/tests/gateway/test_email_imap_id_protocol.py index c44e9c9dbb..9a103f384c 100644 --- a/tests/gateway/test_email_imap_id_protocol.py +++ b/tests/gateway/test_email_imap_id_protocol.py @@ -4,7 +4,6 @@ No external mail service or credentials are used. Unsupported ID gets BAD followed by BYE: swallowing the ID error still leaves SELECT unable to proceed. """ -import asyncio import imaplib import socketserver import threading @@ -80,8 +79,7 @@ def imap_peer(id_mode): @pytest.mark.parametrize("id_mode", ["absent", "accept", "reject"]) -@pytest.mark.parametrize("phase", ["startup", "poll"]) -def test_id_negotiation_preserves_inbox_connection(monkeypatch, id_mode, phase): +def test_id_negotiation_preserves_inbox_connection(monkeypatch, id_mode): from gateway.config import PlatformConfig from plugins.platforms.email.adapter import EmailAdapter @@ -104,18 +102,8 @@ def test_id_negotiation_preserves_inbox_connection(monkeypatch, id_mode, phase): lambda *args, **kwargs: imaplib.IMAP4(*address, timeout=5), ) - if phase == "startup": - - async def connect_and_stop(): - try: - return await adapter.connect() - finally: - await adapter.disconnect() - - assert asyncio.run(connect_and_stop()) is True - else: - assert adapter._fetch_new_messages() == [] - assert adapter._last_fetch_failed is False + assert adapter._fetch_new_messages() == [] + assert adapter._last_fetch_failed is False assert commands.count("ID") == (0 if id_mode == "absent" else 1) assert commands.index("LOGIN") < commands.index("SELECT") < commands.index("UID") From cd71ee070898fe5c4d10c3f5ac36c8a9ca5369e3 Mon Sep 17 00:00:00 2001 From: Benjamin Brumbaugh Date: Sun, 6 Sep 2026 09:00:00 +0000 Subject: [PATCH 168/276] fix(compression): defer local preflight after native checkpoint A native Responses compaction checkpoint is opaque ciphertext; the rough preflight estimator counts it as text (5.17M chars -> ~1.29M tokens against a 204K trigger) and fires local compression on a request whose real prompt is ~116K. Arm the existing one-response real-usage latch when a replayable checkpoint is captured (build_assistant_message) or restored into a fresh agent (_hydrate_from_history), honor it in the post-tool gate and idle compaction, and require non-empty encrypted_content for a checkpoint. Squash of the author's source commits from #100642 (0e3c234ea0, 771e1b3365, bb1505a119) plus the fdf140c81d test refresh, re-based onto current main by patch application. Source delta is byte-identical to the PR head d6ce3e236d. Fixes #100611 --- agent/chat_completion_helpers.py | 13 ++ agent/codex_responses_adapter.py | 46 +++-- agent/context_compressor.py | 17 +- agent/native_compaction.py | 7 +- agent/turn_context.py | 28 ++- agent/turn_context_compaction.py | 5 + agent/turn_preflight.py | 3 + tests/agent/test_context_compressor.py | 34 ++++ tests/run_agent/test_413_compression.py | 168 +++++++++++++++++- .../test_post_tool_compression_attempt_cap.py | 9 + tests/run_agent/test_run_agent.py | 74 ++++++++ 11 files changed, 387 insertions(+), 17 deletions(-) diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index f3952ae1cf..7b0a5022c3 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -1729,6 +1729,19 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic value = getattr(assistant_message, attr, None) if value: msg[attr] = value + if attr == "codex_reasoning_items": + from agent.codex_responses_adapter import ( + has_replayable_native_compaction_checkpoint, + ) + + note_checkpoint = getattr( + agent.context_compressor, "note_native_compaction_checkpoint", None + ) + if ( + callable(note_checkpoint) + and has_replayable_native_compaction_checkpoint(agent, [msg]) + ): + note_checkpoint() if assistant_tool_calls: msg["tool_calls"] = [_assistant_tool_call_dict(agent, tc, i) for i, tc in enumerate(assistant_tool_calls)] diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 5f4395b268..bc079c2d32 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -541,16 +541,10 @@ def classify_responses_route(agent: Any) -> ResponsesRouteFlags: ) -def estimate_native_responses_preflight_tokens( - agent: Any, messages: List[Dict[str, Any]], *, system_prompt: str = "", tools: Optional[List[Dict[str, Any]]] = None, -) -> Optional[int]: - """Estimate tokens for the checkpoint-pruned Responses payload (the full transcript overstates a natively compacted - session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails. - - Automatic preflight previously counted the full durable transcript. On a natively compacted Codex - session that overstates the wire by several times and fires local compression against history the main - request will never send (#96155). - """ +def _native_responses_replay_items( + agent: Any, messages: List[Dict[str, Any]] +) -> Optional[List[Dict[str, Any]]]: + """Build the native-compaction-eligible wire items, or ``None`` when ineligible.""" if getattr(agent, "api_mode", None) != "codex_responses" or not isinstance(messages, list): return None route = classify_responses_route(agent)._asdict() @@ -565,7 +559,37 @@ def estimate_native_responses_preflight_tokens( native_compaction_eligible=True, ) except Exception: - logger.debug("native Responses preflight conversion failed; falling back to generic estimate", exc_info=True) + logger.debug( + "native Responses replay conversion failed; using the generic fallback", + exc_info=True, + ) + return None + return items + + +def has_replayable_native_compaction_checkpoint( + agent: Any, messages: List[Dict[str, Any]] +) -> bool: + """Whether the current route would replay a persisted native checkpoint.""" + items = _native_responses_replay_items(agent, messages) + if items is None: + return False + from agent.native_compaction import has_compaction_checkpoint + return has_compaction_checkpoint(items) + + +def estimate_native_responses_preflight_tokens( + agent: Any, messages: List[Dict[str, Any]], *, system_prompt: str = "", tools: Optional[List[Dict[str, Any]]] = None, +) -> Optional[int]: + """Estimate tokens for the checkpoint-pruned Responses payload (the full transcript overstates a natively compacted + session and fires local compression needlessly). None when native compaction is not proven eligible or conversion fails. + + Automatic preflight previously counted the full durable transcript. On a natively compacted Codex + session that overstates the wire by several times and fires local compression against history the main + request will never send (#96155). + """ + items = _native_responses_replay_items(agent, messages) + if items is None: return None from agent.model_metadata import estimate_request_tokens_rough return estimate_request_tokens_rough(items, system_prompt=system_prompt or "", tools=tools) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 33838ca390..e9242fa371 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -2416,6 +2416,20 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): except (TypeError, ValueError): self._pending_request_rough_tokens = 0 + def note_native_compaction_checkpoint(self) -> None: + """Wait for real usage before trusting a newly checkpointed request. + + Native Responses compaction replaces durable history with an opaque + encrypted checkpoint. Its serialized size is unrelated to the token + count billed by the provider, so the first rough estimate after capture + can jump by more than the whole context window. Reuse the one-response + compaction latch and discard any stale local-compression baseline; the + next provider response then pairs its real usage with the rough estimate + for the checkpointed request. + """ + self.awaiting_real_usage_after_compression = True + self.last_compression_rough_tokens = 0 + def should_defer_preflight_to_real_usage(self, rough_tokens: int) -> bool: """Return True when a high rough preflight estimate is known-noisy. Projects real usage as ``last_real + (rough_now - rough_at_last_real)`` and fires only when the @@ -2426,7 +2440,8 @@ class ContextCompressor(MicroCompactionMixin, ContextEngine): re-runs with the aligned basis.""" if rough_tokens < self.threshold_tokens: return False - # After compaction last_real_prompt_tokens is STALE (above threshold); defer one turn until real usage arrives. + # After local or native compaction, last_real_prompt_tokens is STALE + # (above threshold); defer one turn until real usage arrives. if self.awaiting_real_usage_after_compression: return True if self.last_real_prompt_tokens <= 0 or self.last_real_prompt_tokens >= self.threshold_tokens: diff --git a/agent/native_compaction.py b/agent/native_compaction.py index 0202770c4d..8707cc91af 100644 --- a/agent/native_compaction.py +++ b/agent/native_compaction.py @@ -328,7 +328,12 @@ def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool: def has_compaction_checkpoint(items: Any) -> bool: """Does this ``codex_reasoning_items`` sidecar carry a compaction checkpoint? A checkpoint is cumulative context living in exactly one place: rewrite/discard the sidecar only after asking.""" - return isinstance(items, list) and any(_is_compaction_item(item) for item in items) + return isinstance(items, list) and any( + _is_compaction_item(item) + and isinstance(item.get("encrypted_content"), str) + and bool(item["encrypted_content"].strip()) + for item in items + ) def merge_interim_reasoning_items(prior_items: Any, new_items: Any) -> List[Dict[str, Any]]: diff --git a/agent/turn_context.py b/agent/turn_context.py index 82bba061a8..82bf522cb4 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -525,13 +525,37 @@ def _stage_turn_user_message( def _hydrate_from_history(agent: Any, conversation_history: Optional[List[Any]]) -> None: - """Hydrate the todo store and per-session nudge counters from persisted history.""" + """Hydrate process-local state from persisted history on the first resumed turn.""" if not conversation_history: return if not agent._todo_store.has_items(): agent._hydrate_todo_store(conversation_history) - # Hydrate per-session nudge counters from persisted history. + # A live native checkpoint arms this latch while its response is captured. A + # restarted agent must recover the same one-response deferral before turn-start + # compression can rewrite the restored opaque checkpoint. Reuse the adapter's + # exact route/issuer/replay filtering and tolerate plugin compressors without the + # optional hook. if agent._user_turn_count == 0: + note_checkpoint = getattr( + getattr(agent, "context_compressor", None), + "note_native_compaction_checkpoint", + None, + ) + if callable(note_checkpoint): + try: + from agent.codex_responses_adapter import ( + has_replayable_native_compaction_checkpoint, + ) + + if has_replayable_native_compaction_checkpoint( + agent, conversation_history + ): + note_checkpoint() + except Exception: + logger.debug( + "restored native checkpoint hydration skipped", exc_info=True + ) + # Hydrate per-session nudge counters from persisted history. prior_user_turns = sum(1 for m in conversation_history if m.get("role") == "user") if prior_user_turns > 0: agent._user_turn_count = prior_user_turns diff --git a/agent/turn_context_compaction.py b/agent/turn_context_compaction.py index a9f8a9b703..171f602048 100644 --- a/agent/turn_context_compaction.py +++ b/agent/turn_context_compaction.py @@ -156,6 +156,11 @@ def _idle_compaction( if _idle_gap < _idle_after: return _compressor = agent.context_compressor + # A live or restored native checkpoint must reach its issuer once so real usage, + # rather than an opaque ciphertext estimate, decides whether local compression is + # still needed. Threshold and post-tool preflight honor the same latch. + if bool(getattr(_compressor, "awaiting_real_usage_after_compression", False)): + return # Route-aware pressure: on compacted native-Codex sessions the durable figure # overstates the wire, so reuse the preflight estimator. _idle_tokens = _tc._preflight_request_tokens( diff --git a/agent/turn_preflight.py b/agent/turn_preflight.py index cc2a5ff011..15d661e53a 100644 --- a/agent/turn_preflight.py +++ b/agent/turn_preflight.py @@ -297,6 +297,9 @@ def compress_after_tool_results( if ( agent.compression_enabled and compression_attempts < max_compression_attempts + and not bool( + getattr(_compressor, "awaiting_real_usage_after_compression", False) + ) and _compressor.should_compress(_real_tokens) ): compression_attempts += 1 diff --git a/tests/agent/test_context_compressor.py b/tests/agent/test_context_compressor.py index 10997fe94a..470e7a9661 100644 --- a/tests/agent/test_context_compressor.py +++ b/tests/agent/test_context_compressor.py @@ -366,6 +366,40 @@ class TestPreflightDeferral: compressor.awaiting_real_usage_after_compression = True assert compressor.should_defer_preflight_to_real_usage(95_000) is True + def test_native_checkpoint_defers_until_provider_usage_reanchors(self, compressor): + """A newly captured native checkpoint is opaque ciphertext, not text. + + Its serialized size can add more than a million rough tokens even when + the provider reports that the checkpoint-pruned request fits. Defer the + local compressor for one request so real usage can establish the new + rough/real calibration instead of immediately summarizing again. + """ + compressor.threshold_tokens = 85_000 + compressor.last_real_prompt_tokens = 60_000 + compressor.last_rough_tokens_when_real_prompt_fit = 70_000 + compressor.last_compression_rough_tokens = 40_000 + + compressor.note_native_compaction_checkpoint() + + encrypted_checkpoint_rough = 1_300_000 + assert compressor.awaiting_real_usage_after_compression is True + assert compressor.last_compression_rough_tokens == 0 + assert ( + compressor.should_defer_preflight_to_real_usage( + encrypted_checkpoint_rough + ) + is True + ) + + compressor.note_request_rough_estimate(encrypted_checkpoint_rough) + compressor.update_from_response({"prompt_tokens": 65_000}) + + assert compressor.awaiting_real_usage_after_compression is False + assert ( + compressor.last_rough_tokens_when_real_prompt_fit + == encrypted_checkpoint_rough + ) + diff --git a/tests/run_agent/test_413_compression.py b/tests/run_agent/test_413_compression.py index 0f54c11a7a..e11c2ddef0 100644 --- a/tests/run_agent/test_413_compression.py +++ b/tests/run_agent/test_413_compression.py @@ -12,11 +12,13 @@ import pytest from types import SimpleNamespace +from typing import Any from unittest.mock import MagicMock, patch from agent.context_compressor import SUMMARY_PREFIX, _DB_PERSISTED_MARKER from agent.conversation_compression import COMPACTION_DONE_STATUS, COMPACTION_STATUS +from hermes_state import SessionDB from run_agent import AIAgent import run_agent @@ -78,8 +80,7 @@ def _make_413_error(*, use_status_code=True, message="Request entity too large") return err -@pytest.fixture() -def agent(): +def _new_test_agent(): with ( patch("model_tools.get_tool_definitions", return_value=_make_tool_defs("web_search")), patch("model_tools.check_toolset_requirements", return_value={}), @@ -104,6 +105,11 @@ def agent(): return a +@pytest.fixture() +def agent(): + return _new_test_agent() + + # --------------------------------------------------------------------------- # Tests # --------------------------------------------------------------------------- @@ -1209,6 +1215,164 @@ class TestPreflightCompression: assert agent.context_compressor._ineffective_compression_count == 2 assert agent.context_compressor._last_compression_savings_pct == 0.0 + @pytest.mark.parametrize( + ("failure_kind", "expected_provider_calls"), + [("interrupt", 0), ("provider_error", 3)], + ) + def test_pending_native_checkpoint_recovers_after_failed_turn( + self, agent, failure_kind, expected_provider_calls + ): + """A failed turn preserves the latch only until a response arrives. + + Clearing it on interrupt/error would expose the next turn to the same + ciphertext-driven false preflight compaction. Keeping it armed does not + park the compressor: the next request bypasses local preflight, reaches + the provider, and real usage consumes the latch. + """ + agent.compression_enabled = True + agent.context_compressor.context_length = 200_000 + agent.context_compressor.threshold_tokens = 130_000 + agent.context_compressor.note_native_compaction_checkpoint() + if failure_kind == "interrupt": + agent._interrupt_requested = True + else: + agent.client.chat.completions.create.side_effect = RuntimeError( + "provider died" + ) + + with ( + patch("agent.turn_context.estimate_request_tokens_rough", return_value=1_300_000), + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + ): + failed = agent.run_conversation("first request") + + if failure_kind == "interrupt": + assert failed["interrupted"] is True + agent._interrupt_requested = False + else: + assert failed["failed"] is True + assert ( + agent.client.chat.completions.create.call_count + == expected_provider_calls + ) + assert agent.context_compressor.awaiting_real_usage_after_compression is True + + agent.client.chat.completions.create.reset_mock() + agent.client.chat.completions.create.side_effect = None + agent.client.chat.completions.create.return_value = _mock_response( + content="Recovered", + usage={ + "prompt_tokens": 65_000, + "completion_tokens": 100, + "total_tokens": 65_100, + }, + ) + + with ( + patch("agent.turn_context.estimate_request_tokens_rough", return_value=1_300_000), + patch.object(agent, "_compress_context") as mock_compress, + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + ): + recovered = agent.run_conversation("retry after failure") + + assert recovered["completed"] is True + assert recovered["final_response"] == "Recovered" + mock_compress.assert_not_called() + assert agent.client.chat.completions.create.call_count == 1 + assert agent.context_compressor.awaiting_real_usage_after_compression is False + assert agent.context_compressor.last_prompt_tokens == 65_000 + + def test_restored_native_checkpoint_defers_first_local_compaction(self, tmp_path): + """A checkpoint restored into a new agent instance must reach its issuer once. + + The in-memory native-checkpoint latch is lost when the process restarts. Rehydrate + it in an agent constructed after the durable history is reopened, before either + idle or threshold preflight can summarize the opaque checkpoint using its + ciphertext-sized rough estimate. + """ + db_path = tmp_path / "state.db" + session_id = "restored-native-checkpoint" + checkpoint = { + "type": "compaction", + "encrypted_content": "opaque-checkpoint", + "_issuer_kind": "codex_backend", + } + db = SessionDB(db_path=db_path) + db.create_session(session_id, source="test") + db.append_message(session_id, "user", "before restart") + db.append_message( + session_id, + "assistant", + "checkpoint captured", + codex_reasoning_items=[checkpoint], + ) + db.close() + + reopened = SessionDB(db_path=db_path) + history = reopened.get_messages_as_conversation(session_id) + reopened.close() + assert history[-1]["codex_reasoning_items"] == [checkpoint] + + agent: Any = _new_test_agent() + assert agent.context_compressor.awaiting_real_usage_after_compression is False + agent.api_mode = "codex_responses" + agent.provider = "openai-codex" + agent.model = "gpt-5.6-sol" + agent.base_url = "https://chatgpt.com/backend-api/codex" + agent._base_url_hostname = "chatgpt.com" + agent._base_url_lower = agent.base_url + agent.codex_responses_native_compaction = True + agent.runtime_capabilities = {"native_compaction": True} + agent.context_compressor.context_length = 200_000 + agent.context_compressor.threshold_tokens = 130_000 + # Exercise the idle pass too: it runs before threshold preflight and must honor + # the same one-response checkpoint latch on a long-idle restored session. + agent.compression_idle_compact_after_seconds = 1 + agent._last_activity_ts = 0 + response = SimpleNamespace( + output=[ + SimpleNamespace( + type="message", + status="completed", + content=[SimpleNamespace(type="output_text", text="Resumed")], + ) + ], + usage=SimpleNamespace( + input_tokens=65_000, + output_tokens=100, + total_tokens=65_100, + ), + status="completed", + incomplete_details=None, + model="gpt-5.6-sol", + ) + + with ( + patch( + "agent.codex_responses_adapter.estimate_native_responses_preflight_tokens", + return_value=1_300_000, + ), + patch.object(agent, "_run_codex_stream", return_value=response) as provider, + patch.object(agent, "_compress_context") as local_compress, + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + ): + resumed = agent.run_conversation( + "after restart", conversation_history=history + ) + + assert resumed["completed"] is True + assert resumed["final_response"] == "Resumed" + local_compress.assert_not_called() + provider.assert_called_once() + assert agent.context_compressor.awaiting_real_usage_after_compression is False + assert agent.context_compressor.last_prompt_tokens == 65_000 + class TestToolResultPreflightCompression: """Compression should trigger when tool results push context past the threshold.""" diff --git a/tests/run_agent/test_post_tool_compression_attempt_cap.py b/tests/run_agent/test_post_tool_compression_attempt_cap.py index b70fd881b9..63d4558a35 100644 --- a/tests/run_agent/test_post_tool_compression_attempt_cap.py +++ b/tests/run_agent/test_post_tool_compression_attempt_cap.py @@ -86,6 +86,7 @@ def _pressured_compressor() -> MagicMock: compressor.threshold_tokens = 10_000 compressor.context_length = 200_000 compressor.last_prompt_tokens = 150_000 + compressor.awaiting_real_usage_after_compression = False compressor.should_compress.return_value = True compressor.should_defer_preflight_to_real_usage.return_value = True compressor.get_active_compression_failure_cooldown.return_value = None @@ -151,6 +152,14 @@ def _run_tool_loop(agent, n_tool_iterations: int): class TestPostToolCompressionAttemptCap: + def test_post_tool_gate_waits_for_usage_after_native_checkpoint(self, agent): + agent.context_compressor.awaiting_real_usage_after_compression = True + + result, compress_calls = _run_tool_loop(agent, n_tool_iterations=1) + + assert result["completed"] is True + assert compress_calls == [] + def test_post_tool_compression_capped_at_default_three(self, agent): """7 tool iterations under constant pressure → exactly 3 compactions. diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 29cd2692f3..438956711f 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -1632,6 +1632,18 @@ class TestBuildApiKwargs: class TestBuildAssistantMessage: + @staticmethod + def _enable_native_compaction(agent): + agent.api_mode = "codex_responses" + agent.provider = "openai-codex" + agent.model = "gpt-5.6-sol" + agent.base_url = "https://chatgpt.com/backend-api/codex" + agent._base_url_hostname = "chatgpt.com" + agent._base_url_lower = agent.base_url + agent.codex_responses_native_compaction = True + agent.compression_enabled = True + agent.runtime_capabilities = {"native_compaction": True} + def test_basic_message(self, agent): msg = _mock_assistant_msg(content="Hello!") result = agent._build_assistant_message(msg, "stop") @@ -1639,6 +1651,68 @@ class TestBuildAssistantMessage: assert result["content"] == "Hello!" assert result["finish_reason"] == "stop" + def test_native_checkpoint_arms_real_usage_preflight_deferral(self, agent): + checkpoint = { + "type": "compaction", + "encrypted_content": "opaque-checkpoint", + "_issuer_kind": "codex_backend", + } + msg = _mock_assistant_msg(content="Compacted") + msg.codex_reasoning_items = [checkpoint] + agent.context_compressor.note_native_compaction_checkpoint = MagicMock() + self._enable_native_compaction(agent) + + result = agent._build_assistant_message(msg, "stop") + + assert result["codex_reasoning_items"] == [checkpoint] + agent.context_compressor.note_native_compaction_checkpoint.assert_called_once_with() + + def test_native_checkpoint_remains_compatible_with_plugin_context_engine(self, agent): + checkpoint = { + "type": "compaction", + "encrypted_content": "opaque-checkpoint", + "_issuer_kind": "codex_backend", + } + msg = _mock_assistant_msg(content="Compacted") + msg.codex_reasoning_items = [checkpoint] + agent.context_compressor = SimpleNamespace(threshold_tokens=204_000) + self._enable_native_compaction(agent) + + result = agent._build_assistant_message(msg, "stop") + + assert result["codex_reasoning_items"] == [checkpoint] + + @pytest.mark.parametrize("encrypted_content", ["", " "]) + def test_malformed_checkpoint_does_not_arm_deferral( + self, agent, encrypted_content + ): + note_checkpoint = MagicMock() + agent.context_compressor.note_native_compaction_checkpoint = note_checkpoint + malformed = { + "type": "compaction", + "encrypted_content": encrypted_content, + } + msg = _mock_assistant_msg(content="Compacted") + msg.codex_reasoning_items = [malformed] + self._enable_native_compaction(agent) + + result = agent._build_assistant_message(msg, "stop") + + assert result["codex_reasoning_items"] == [malformed] + note_checkpoint.assert_not_called() + + def test_ineligible_route_checkpoint_does_not_arm_deferral(self, agent): + note_checkpoint = MagicMock() + agent.context_compressor.note_native_compaction_checkpoint = note_checkpoint + checkpoint = {"type": "compaction", "encrypted_content": "opaque-checkpoint"} + msg = _mock_assistant_msg(content="Compacted") + msg.codex_reasoning_items = [checkpoint] + + result = agent._build_assistant_message(msg, "stop") + + assert result["codex_reasoning_items"] == [checkpoint] + note_checkpoint.assert_not_called() + From 5da6dcda5acd82e3ee8c86d4131c87ad2079f30a Mon Sep 17 00:00:00 2001 From: GodsBoy Date: Sat, 5 Sep 2026 17:56:53 +0200 Subject: [PATCH 169/276] test(compression): align checkpoint fixture and contributor mapping --- contributors/emails/benbrumbaugh@gmail.com | 1 + tests/run_agent/test_proactive_prune_loop_wiring.py | 1 + 2 files changed, 2 insertions(+) create mode 100644 contributors/emails/benbrumbaugh@gmail.com diff --git a/contributors/emails/benbrumbaugh@gmail.com b/contributors/emails/benbrumbaugh@gmail.com new file mode 100644 index 0000000000..cadd027dd5 --- /dev/null +++ b/contributors/emails/benbrumbaugh@gmail.com @@ -0,0 +1 @@ +benjaminbrumbaugh diff --git a/tests/run_agent/test_proactive_prune_loop_wiring.py b/tests/run_agent/test_proactive_prune_loop_wiring.py index 435b55515b..374f81d030 100644 --- a/tests/run_agent/test_proactive_prune_loop_wiring.py +++ b/tests/run_agent/test_proactive_prune_loop_wiring.py @@ -83,6 +83,7 @@ def _quiet_compressor() -> MagicMock: compressor.threshold_tokens = 500_000 compressor.context_length = 1_000_000 compressor.last_prompt_tokens = 120_000 + compressor.awaiting_real_usage_after_compression = False compressor.should_compress.return_value = False compressor.should_compress_info.return_value = (False, None) compressor.should_defer_preflight_to_real_usage.return_value = True From e4a86ec9edf483f368760455b329b89084a7d718 Mon Sep 17 00:00:00 2001 From: GodsBoy Date: Sat, 5 Sep 2026 18:16:16 +0200 Subject: [PATCH 170/276] test(compression): initialise idle checkpoint deferral state --- tests/agent/test_idle_compaction_lock_and_guards.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/agent/test_idle_compaction_lock_and_guards.py b/tests/agent/test_idle_compaction_lock_and_guards.py index d1c30a024a..4426a8804a 100644 --- a/tests/agent/test_idle_compaction_lock_and_guards.py +++ b/tests/agent/test_idle_compaction_lock_and_guards.py @@ -45,6 +45,7 @@ def _prep_idle_agent(db: SessionDB, session_id: str, *, idle_after: int = 60, agent.context_compressor.summary_target_ratio = 0.20 agent.context_compressor.protect_first_n = 2 agent.context_compressor.protect_last_n = 2 + agent.context_compressor.awaiting_real_usage_after_compression = False # No active failure cooldown unless a test installs one. agent.context_compressor.get_active_compression_failure_cooldown = ( lambda *a, **k: None From 8d4b7f874d59841394536c72445bf7d0c6c18f2c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 02:42:41 -0700 Subject: [PATCH 171/276] eval(native-compaction): real-AIAgent A/B probe for checkpoint preflight false-trigger Fake local Responses SSE server plays the Codex backend; scenarios cover live capture (CLI same-object history and gateway reloaded history), SessionDB close/reopen restore into a fresh agent, and the over-threshold negative where local compression must still fire after real usage arrives. --- .../ab_checkpoint_preflight.py | 270 ++++++++++++++++++ 1 file changed, 270 insertions(+) create mode 100644 evals/native_compaction/ab_checkpoint_preflight.py diff --git a/evals/native_compaction/ab_checkpoint_preflight.py b/evals/native_compaction/ab_checkpoint_preflight.py new file mode 100644 index 0000000000..dd17829379 --- /dev/null +++ b/evals/native_compaction/ab_checkpoint_preflight.py @@ -0,0 +1,270 @@ +"""A/B probe: does a freshly captured / restored native compaction checkpoint false-trigger +local compression on the next preflight? (#100611) + +Runs the REAL ``AIAgent`` turn loop (``run_conversation``) against a local fake OpenAI +Responses SSE server that plays the ChatGPT Codex backend role (``provider="openai-codex"`` +→ ``api_mode="codex_responses"``, ``is_codex_backend=True``). No mocks on the agent path +except a counting wrapper around ``_compress_context`` (a real summarizer call would need +a second LLM; the question under test is *whether it fires*, not what it writes). + +Scenarios (all deterministic, no network beyond 127.0.0.1): + +1. ``capture`` — turn 1 returns a ``compaction`` output item carrying N chars of + ciphertext plus real usage below threshold; turn 2 in the SAME agent must reach the + provider without local compression. +2. ``restore`` — the turn-1 transcript is written to a real ``SessionDB``, the DB is + closed/reopened, a FRESH ``AIAgent`` resumes it; its first turn must reach the provider + without local compression (idle pass armed too). +3. ``over_threshold`` (negative) — same as ``capture`` but the provider's real usage after + the checkpoint is ABOVE the local threshold; local compression MUST still fire once real + usage arrives (the deferral is one request, not a disable). + +Usage (from a checkout root, venv python):: + + python evals/native_compaction/ab_checkpoint_preflight.py --out /tmp/result.json +""" + +from __future__ import annotations + +import argparse +import json +import os +import sys +import tempfile +import threading +import time +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(ROOT)) + +THRESHOLD = 204_000 +CONTEXT_LENGTH = 400_000 +# Reported field figure (#100611): 5,169,420 ciphertext chars → ~1.29M rough tokens. +CHECKPOINT_CHARS = 5_169_420 + + +class _FakeResponses: + """Local Responses API: every POST /responses answers one scripted SSE response.""" + + def __init__(self) -> None: + self.requests: list[dict] = [] + self.script: list[dict] = [] + self.lock = threading.Lock() + server = self + + class Handler(BaseHTTPRequestHandler): + def log_message(self, *_a): # noqa: D401 + pass + + def do_POST(self): + n = int(self.headers.get("content-length", 0)) + body = json.loads(self.rfile.read(n) or b"{}") + if not self.path.rstrip("/").endswith("/responses"): + self.send_response(404) + self.end_headers() + return + with server.lock: + server.requests.append(body) + scripted = server.script.pop(0) if server.script else _text_response("ok", 1000) + self.send_response(200) + self.send_header("content-type", "text/event-stream") + self.end_headers() + events = [ + {"type": "response.output_item.done", "output_index": i, "item": item} + for i, item in enumerate(scripted["output"]) + ] + [{"type": "response.completed", "response": scripted}] + for ev in events: + self.wfile.write(f"data: {json.dumps(ev)}\n\n".encode()) + self.wfile.write(b"data: [DONE]\n\n") + self.wfile.flush() + + self.server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + threading.Thread(target=self.server.serve_forever, daemon=True).start() + self.base_url = f"http://127.0.0.1:{self.server.server_address[1]}/backend-api/codex" + + def close(self) -> None: + self.server.shutdown() + self.server.server_close() + + +def _text_response(text: str, input_tokens: int, *, compaction_chars: int = 0) -> dict: + output = [] + if compaction_chars: + output.append({"type": "compaction", "id": "cmp_1", "encrypted_content": "Z" * compaction_chars}) + output.append({ + "type": "message", "id": "msg_1", "role": "assistant", "status": "completed", + "content": [{"type": "output_text", "text": text, "annotations": []}], + }) + return { + "id": "resp_1", "object": "response", "created_at": 0, "status": "completed", + "model": "gpt-5.6", "output": output, + "usage": {"input_tokens": input_tokens, "output_tokens": 10, "total_tokens": input_tokens + 10}, + } + + +def _make_agent(base_url: str, session_id: str | None = None): + from run_agent import AIAgent + + agent = AIAgent( + api_key="test-key", base_url=base_url, provider="openai-codex", model="gpt-5.6", + quiet_mode=True, skip_context_files=True, skip_memory=True, enabled_toolsets=[], + max_iterations=3, session_id=session_id, + ) + agent.compression_enabled = True + agent.codex_responses_native_compaction = True + cc = agent.context_compressor + cc.context_length = CONTEXT_LENGTH + cc.threshold_tokens = THRESHOLD + calls: list[int] = [] + original = agent._compress_context + + def counting(messages, system_message, **kw): + calls.append(int(kw.get("approx_tokens") or 0)) + return messages, kw.get("active_system_prompt") or (system_message.get("content") if isinstance(system_message, dict) else system_message) + + agent._compress_context = counting # type: ignore[method-assign] + agent._ab_compress_calls = calls + agent._ab_original_compress = original + return agent + + +def _request_facts(req: dict) -> dict: + inp = req.get("input") or [] + return { + "context_management": req.get("context_management"), + "input_items": len(inp), + "replayed_compaction_items": sum(1 for i in inp if isinstance(i, dict) and i.get("type") == "compaction"), + "replayed_compaction_chars": sum(len(i.get("encrypted_content") or "") for i in inp if isinstance(i, dict) and i.get("type") == "compaction"), + } + + +def _preflight_estimate(agent, messages) -> int | None: + from agent.codex_responses_adapter import estimate_native_responses_preflight_tokens + + return estimate_native_responses_preflight_tokens(agent, messages, system_prompt="", tools=None) + + +def scenario_capture(wire: _FakeResponses, *, usage_after: int, reload_history: bool) -> dict: + """``reload_history=True`` models the gateway: history is re-read from the DB before + every turn, so message dicts are fresh objects and the usage anchor (keyed on ``id``) + is stale — the rough estimator decides. ``False`` is the CLI shape (anchor protects).""" + wire.requests.clear() + wire.script[:] = [ + _text_response("checkpointed", 63_474, compaction_chars=CHECKPOINT_CHARS), + _text_response("second", usage_after), + _text_response("third", usage_after), + ] + agent = _make_agent(wire.base_url) + r1 = agent.run_conversation("first request") + history = r1["messages"] + if reload_history: + history = json.loads(json.dumps(history)) + carrier = next((m for m in history if m.get("role") == "assistant" and m.get("codex_reasoning_items")), None) + est = _preflight_estimate(agent, history) + latch_after_t1 = bool(agent.context_compressor.awaiting_real_usage_after_compression) + compress_before_t2 = len(agent._ab_compress_calls) + r2 = agent.run_conversation("second request", conversation_history=history) + compress_t2 = len(agent._ab_compress_calls) - compress_before_t2 + history3 = r2["messages"] + if reload_history: + history3 = json.loads(json.dumps(history3)) + r3 = agent.run_conversation("third request", conversation_history=history3) + return { + "turn1_completed": bool(r1.get("completed")), + "checkpoint_persisted": bool(carrier), + "checkpoint_chars": len(carrier["codex_reasoning_items"][0]["encrypted_content"]) if carrier else 0, + "preflight_estimate_before_turn2": est, + "threshold": THRESHOLD, + "latch_armed_after_turn1": latch_after_t1, + "turn2_completed": bool(r2.get("completed")), + "local_compress_calls_turn2": compress_t2, + "turn3_completed": bool(r3.get("completed")), + "local_compress_calls_turn3": len(agent._ab_compress_calls) - compress_before_t2 - compress_t2, + "local_compress_approx_tokens": list(agent._ab_compress_calls), + "provider_requests_total": len(wire.requests), + "requests": [_request_facts(r) for r in wire.requests], + "latch_after_turn3": bool(agent.context_compressor.awaiting_real_usage_after_compression), + "last_real_prompt_tokens": agent.context_compressor.last_real_prompt_tokens, + } + + +def scenario_restore(wire: _FakeResponses, tmp: Path) -> dict: + from hermes_state import SessionDB + + wire.requests.clear() + wire.script[:] = [ + _text_response("checkpointed", 63_474, compaction_chars=CHECKPOINT_CHARS), + _text_response("resumed", 115_802), + ] + sid = "ab-native-restore" + agent = _make_agent(wire.base_url, session_id=sid) + r1 = agent.run_conversation("first request") + db_path = tmp / "state.db" + db = SessionDB(db_path=db_path) + db.create_session(sid, source="cli") + for m in r1["messages"]: + if m.get("role") not in ("user", "assistant", "tool"): + continue + extra = {k: m[k] for k in ("codex_reasoning_items",) if m.get(k)} + db.append_message(sid, m["role"], m.get("content") or "", **extra) + db.close() + reopened = SessionDB(db_path=db_path) + history = reopened.get_messages_as_conversation(sid) + reopened.close() + restored_carrier = next((m for m in history if m.get("codex_reasoning_items")), None) + fresh = _make_agent(wire.base_url, session_id=sid) + # Idle pass armed: a long-idle restored session runs _idle_compaction before threshold preflight. + fresh.compression_idle_compact_after_seconds = 1 + fresh._last_activity_ts = time.time() - 3600 + est = _preflight_estimate(fresh, history) + n_req_before = len(wire.requests) + r2 = fresh.run_conversation("after restart", conversation_history=history) + return { + "turn1_completed": bool(r1.get("completed")), + "restored_checkpoint_chars": len(restored_carrier["codex_reasoning_items"][0]["encrypted_content"]) if restored_carrier else 0, + "preflight_estimate_fresh_agent": est, + "threshold": THRESHOLD, + "resume_completed": bool(r2.get("completed")), + "local_compress_calls_resume": len(fresh._ab_compress_calls), + "local_compress_approx_tokens": list(fresh._ab_compress_calls), + "provider_requests_resume": len(wire.requests) - n_req_before, + "requests": [_request_facts(r) for r in wire.requests[n_req_before:]], + "latch_after_resume": bool(fresh.context_compressor.awaiting_real_usage_after_compression), + "last_real_prompt_tokens": fresh.context_compressor.last_real_prompt_tokens, + } + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--out", required=True) + args = ap.parse_args() + tmp = Path(tempfile.mkdtemp(prefix="ab-native-")) + os.environ["HERMES_HOME"] = str(tmp / "home") + (tmp / "home").mkdir(parents=True) + import subprocess + + head = subprocess.run(["git", "rev-parse", "HEAD"], cwd=ROOT, capture_output=True, text=True).stdout.strip() + wire = _FakeResponses() + try: + result = { + "checkout": str(ROOT), "head": head, + "capture_cli_same_objects": scenario_capture(wire, usage_after=115_802, reload_history=False), + "capture_gateway_reloaded_history": scenario_capture(wire, usage_after=115_802, reload_history=True), + "restore": scenario_restore(wire, tmp), + # Negative: real usage after the checkpoint is STILL over threshold → local + # compression must fire on the following turn (deferral is one request, not a disable). + "over_threshold_negative": scenario_capture(wire, usage_after=THRESHOLD + 5_000, reload_history=True), + } + finally: + wire.close() + Path(args.out).write_text(json.dumps(result, indent=2, default=str), encoding="utf-8") + print(json.dumps({k: (v if not isinstance(v, dict) else { + kk: vv for kk, vv in v.items() if kk not in ("requests",) + }) for k, v in result.items()}, indent=2, default=str)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) From 9186e3ebc5cdb2c3785f4e26cc7ad57b9c2d9d52 Mon Sep 17 00:00:00 2001 From: Adolanium <94890352+Adolanium@users.noreply.github.com> Date: Sat, 5 Sep 2026 20:20:40 +0300 Subject: [PATCH 172/276] feat(desktop): session import view for foreign coding-agent transcripts Browse the backend host's foreign CLI session logs, preview a bounded read-only transcript, and continue a copy in Hermes under the selected profile. Reuses the hermes_cli.foreign_sessions parsers and the portability validator/writer; imports are transactional and deduplicated on the recorded origin. --- apps/desktop/src/app/chat/sidebar/index.tsx | 11 +- .../desktop/src/app/command-palette/index.tsx | 8 + apps/desktop/src/app/contrib/wiring.tsx | 13 + apps/desktop/src/app/routes.ts | 5 + apps/desktop/src/app/session-import/api.ts | 49 +++ .../src/app/session-import/index.test.tsx | 102 ++++++ apps/desktop/src/app/session-import/index.tsx | 320 ++++++++++++++++++ .../components/assistant-ui/markdown-text.tsx | 16 +- apps/desktop/src/i18n/ar.ts | 36 ++ apps/desktop/src/i18n/en.ts | 37 ++ apps/desktop/src/i18n/ja.ts | 36 ++ apps/desktop/src/i18n/ru.ts | 37 ++ apps/desktop/src/i18n/types.ts | 36 ++ apps/desktop/src/i18n/zh-hant.ts | 36 ++ apps/desktop/src/i18n/zh.ts | 36 ++ hermes_cli/foreign_session_browser.py | 111 ++++++ hermes_state_portability.py | 49 +++ tests/tui_gateway/test_session_foreign.py | 97 ++++++ tui_gateway/methods_session_foreign.py | 46 +++ tui_gateway/server.py | 5 +- 20 files changed, 1077 insertions(+), 9 deletions(-) create mode 100644 apps/desktop/src/app/session-import/api.ts create mode 100644 apps/desktop/src/app/session-import/index.test.tsx create mode 100644 apps/desktop/src/app/session-import/index.tsx create mode 100644 hermes_cli/foreign_session_browser.py create mode 100644 tests/tui_gateway/test_session_foreign.py create mode 100644 tui_gateway/methods_session_foreign.py diff --git a/apps/desktop/src/app/chat/sidebar/index.tsx b/apps/desktop/src/app/chat/sidebar/index.tsx index 3f53fafa7f..0aefa90e30 100644 --- a/apps/desktop/src/app/chat/sidebar/index.tsx +++ b/apps/desktop/src/app/chat/sidebar/index.tsx @@ -3,9 +3,10 @@ import { sortableKeyboardCoordinates } from '@dnd-kit/sortable' import { useStore } from '@nanostores/react' import type * as React from 'react' import { useCallback, useEffect, useMemo, useRef, useState } from 'react' -import { useLocation } from 'react-router' +import { useLocation, useNavigate } from 'react-router' import { PlatformAvatar } from '@/app/messaging/platform-icon' +import { SESSION_IMPORT_ROUTE } from '@/app/routes' import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' import { ContextMenu, ContextMenuContent, ContextMenuTrigger } from '@/components/ui/context-menu' @@ -333,6 +334,7 @@ export function ChatSidebar({ const { t } = useI18n() const s = t.sidebar const { pathname } = useLocation() + const navigate = useNavigate() // Contributed nav rows (plugins pairing a page with a sidebar entry) render // below the built-ins with the same chrome; active = at their route. const navContributions = useContributions(SIDEBAR_NAV_AREA) @@ -1617,6 +1619,13 @@ export function ChatSidebar({ +
+ +
+ {showSessionSections && (
void }) { label: cc.sections.sessions, run: go(`${COMMAND_CENTER_ROUTE}?section=sessions`) }, + { + icon: Download, + id: 'session-import', + keywords: ['import', 'claude', 'codex', 'conversation'], + label: t.sessionImport.action, + run: go(SESSION_IMPORT_ROUTE) + }, { icon: Activity, id: 'cc-system', diff --git a/apps/desktop/src/app/contrib/wiring.tsx b/apps/desktop/src/app/contrib/wiring.tsx index 0783e243ac..2e7881845e 100644 --- a/apps/desktop/src/app/contrib/wiring.tsx +++ b/apps/desktop/src/app/contrib/wiring.tsx @@ -113,6 +113,7 @@ import { SETTINGS_ROUTE, syncWorkspaceRoute } from '../routes' +import { SessionImportView } from '../session-import' import { SessionPickerOverlay } from '../session-picker-overlay' import { SessionSwitcher } from '../session-switcher' import { useBackgroundQueueDrain } from '../session/hooks/use-background-queue-drain' @@ -1246,6 +1247,18 @@ export function ContribWiring({ children }: { children: ReactNode }) { )} + {currentView === 'session-import' && ( + { + closeOverlayToPreviousRoute() + openSession(sessionId, navigate, 'stack') + }} + owner={{ connectionId: activeConnectionId || 'local', profile: activeGatewayProfile }} + /> + )} + {commandCenterOpen && ( = new Set([ + 'session-import', 'agents', 'command-center', 'cron', diff --git a/apps/desktop/src/app/session-import/api.ts b/apps/desktop/src/app/session-import/api.ts new file mode 100644 index 0000000000..13fe71464b --- /dev/null +++ b/apps/desktop/src/app/session-import/api.ts @@ -0,0 +1,49 @@ +import { requestGatewayForAgent } from '@/store/gateway' +import type { SessionOwnerRoute } from '@/store/session-request-router' + +export interface ForeignSession { + id: string + source: 'claude' | 'codex' + label: string + title: string + cwd: string | null + mtime: number + turn_count: number + excerpt: string +} + +export interface ForeignPage { + sessions: ForeignSession[] + next_offset: number | null + host: string + unreadable: number +} + +export interface ForeignPreview { + truncated: boolean + messages: { role: string; content: string }[] + total: number + already_imported: string | null + cwd: string | null +} + +export interface ForeignImportResult { + session_id: string + already_imported: boolean +} + +export function foreignRequest( + owner: SessionOwnerRoute, + method: 'list' | 'preview' | 'import', + params: Record, + signal?: AbortSignal +) { + return requestGatewayForAgent( + owner.connectionId, + owner.profile, + `session.foreign.${method}`, + params, + 60_000, + signal + ) +} diff --git a/apps/desktop/src/app/session-import/index.test.tsx b/apps/desktop/src/app/session-import/index.test.tsx new file mode 100644 index 0000000000..2656245bd5 --- /dev/null +++ b/apps/desktop/src/app/session-import/index.test.tsx @@ -0,0 +1,102 @@ +import { QueryClient, QueryClientProvider } from '@tanstack/react-query' +import { cleanup, fireEvent, render, screen, waitFor } from '@testing-library/react' +import { afterEach, expect, it, vi } from 'vitest' + +import { setSessionOwnerHint } from '@/store/session' + +import { type ForeignImportResult, foreignRequest } from './api' + +import { SessionImportView } from './index' + +vi.mock('./api', () => ({ foreignRequest: vi.fn() })) +vi.mock('@/store/session', () => ({ setSessionOwnerHint: vi.fn() })) +vi.mock('@/components/assistant-ui/markdown-text', () => ({ + MarkdownTextContent: ({ text }: { text: string }) =>

{text}

+})) +vi.mock('../overlays/overlay-view', () => ({ + OverlayView: ({ children }: { children: React.ReactNode }) =>
{children}
+})) + +afterEach(() => { + cleanup() + vi.clearAllMocks() +}) + +const owner = { connectionId: 'workstation', profile: 'research' } + +const session = { + id: 'foreign-one', + source: 'claude', + label: 'Claude Code', + title: 'Repair imports', + cwd: '/work/project', + mtime: 1000, + turn_count: 2, + excerpt: 'Help with imports' +} + +function mount(onOpenSession = vi.fn()) { + return { + onOpenSession, + ...render( + + + + ) + } +} + +it('browses without importing, then retries a failed import on the captured owner before opening', async () => { + let attempts = 0 + vi.mocked(foreignRequest).mockImplementation(async (_owner, method) => { + if (method === 'list') { + return { sessions: [session], next_offset: null, host: 'studio', unreadable: 0 } + } + + if (method === 'preview') { + return { messages: [{ role: 'user', content: 'Please repair this' }], total: 1, already_imported: null } + } + + if (++attempts === 1) { + throw new Error('Connection interrupted') + } + + return { session_id: 'durable-one', already_imported: false } + }) + const { onOpenSession } = mount() + fireEvent.click(await screen.findByRole('button', { name: /Repair imports/ })) + await screen.findByText('Please repair this') + expect(attempts).toBe(0) + fireEvent.click(screen.getByRole('button', { name: 'Continue in Hermes' })) + await screen.findByRole('alert') + expect(onOpenSession).not.toHaveBeenCalled() + fireEvent.click(screen.getByRole('button', { name: 'Continue in Hermes' })) + await waitFor(() => expect(onOpenSession).toHaveBeenCalledWith('durable-one')) + expect(setSessionOwnerHint).toHaveBeenCalledWith('durable-one', owner) + expect(foreignRequest).toHaveBeenCalledWith(owner, 'import', { id: 'foreign-one' }, expect.any(AbortSignal)) +}) + +it('does not navigate when an import finishes after the view has closed', async () => { + let finish!: (result: ForeignImportResult) => void + vi.mocked(foreignRequest).mockImplementation(async (_owner, method) => { + if (method === 'list') { + return { sessions: [session], next_offset: null, host: 'studio', unreadable: 0 } + } + + if (method === 'preview') { + return { messages: [], total: 0, already_imported: 'existing' } + } + + return new Promise(resolve => { + finish = resolve + }) + }) + const { onOpenSession, unmount } = mount() + fireEvent.click(await screen.findByRole('button', { name: /Repair imports/ })) + fireEvent.click(await screen.findByRole('button', { name: 'Open in Hermes' })) + await waitFor(() => expect(finish).toBeDefined()) + unmount() + finish({ session_id: 'existing', already_imported: true }) + await waitFor(() => expect(setSessionOwnerHint).toHaveBeenCalledWith('existing', owner)) + expect(onOpenSession).not.toHaveBeenCalled() +}) diff --git a/apps/desktop/src/app/session-import/index.tsx b/apps/desktop/src/app/session-import/index.tsx new file mode 100644 index 0000000000..47eef212b4 --- /dev/null +++ b/apps/desktop/src/app/session-import/index.tsx @@ -0,0 +1,320 @@ +import { useInfiniteQuery, useQuery, useQueryClient } from '@tanstack/react-query' +import { useEffect, useRef, useState } from 'react' + +import { MarkdownTextContent } from '@/components/assistant-ui/markdown-text' +import { Button } from '@/components/ui/button' +import { Codicon } from '@/components/ui/codicon' +import { EmptyState } from '@/components/ui/empty-state' +import { ErrorState } from '@/components/ui/error-state' +import { Loader } from '@/components/ui/loader' +import { SearchField } from '@/components/ui/search-field' +import { SegmentedControl } from '@/components/ui/segmented-control' +import { useI18n } from '@/i18n' +import { cn } from '@/lib/utils' +import { setSessionOwnerHint } from '@/store/session' +import type { SessionOwnerRoute } from '@/store/session-request-router' + +import { OverlayView } from '../overlays/overlay-view' +import { PanelEmpty } from '../overlays/panel' + +import { type ForeignImportResult, type ForeignPage, type ForeignPreview, foreignRequest } from './api' + +interface SessionImportViewProps { + owner: SessionOwnerRoute + onClose: () => void + onOpenSession: (id: string) => void +} + +export function SessionImportView({ owner, onClose, onOpenSession }: SessionImportViewProps) { + const { t, locale } = useI18n() + const copy = t.sessionImport + const queryClient = useQueryClient() + const [source, setSource] = useState<'all' | 'claude' | 'codex'>('all') + const [search, setSearch] = useState('') + const [selected, setSelected] = useState(null) + const [pending, setPending] = useState(false) + const [error, setError] = useState('') + const lifetime = useRef(null) + // eslint-disable-next-line no-restricted-syntax -- lifetime cancellation, not a mirrored reactive value + useEffect(() => { + const controller = new AbortController() + lifetime.current = controller + + return () => controller.abort() + }, []) + const scope = [owner.connectionId, owner.profile] + + const sessions = useInfiniteQuery({ + queryKey: ['foreign-sessions', ...scope, source], + initialPageParam: 0, + queryFn: ({ pageParam, signal }) => + foreignRequest( + owner, + 'list', + { + source: source === 'all' ? null : source, + offset: pageParam + }, + signal + ), + getNextPageParam: page => page.next_offset, + retry: false + }) + + const rows = [ + ...new Map((sessions.data?.pages.flatMap(page => page.sessions) ?? []).map(row => [row.id, row])).values() + ] + + const visible = rows.filter(row => + `${row.title} ${row.cwd ?? ''} ${row.excerpt}`.toLocaleLowerCase().includes(search.toLocaleLowerCase()) + ) + + const current = rows.find(row => row.id === selected) + + const preview = useQuery({ + queryKey: ['foreign-preview', ...scope, selected], + queryFn: ({ signal }) => foreignRequest(owner, 'preview', { id: selected }, signal), + enabled: Boolean(current), + retry: false + }) + + const host = sessions.data?.pages[0]?.host + const unreadable = sessions.data?.pages.reduce((sum, page) => sum + page.unreadable, 0) ?? 0 + + async function continueSession() { + if (!current || pending) { + return + } + + const signal = lifetime.current?.signal + setPending(true) + setError('') + + try { + const result = await foreignRequest(owner, 'import', { id: current.id }, signal) + setSessionOwnerHint(result.session_id, owner) + void queryClient.invalidateQueries({ queryKey: ['foreign-preview', ...scope] }) + void queryClient.invalidateQueries({ queryKey: ['sessions'] }) + + if (!signal?.aborted) { + onOpenSession(result.session_id) + } + } catch (cause) { + if (!signal?.aborted) { + setError(cause instanceof Error ? cause.message : copy.importError) + } + } finally { + if (!signal?.aborted) { + setPending(false) + } + } + } + + return ( + { + void sessions.refetch() + + if (current) { + void preview.refetch() + } + }} + size="icon-titlebar" + variant="ghost" + > + + + } + > +
+
+

{copy.title}

+

{copy.subtitle}

+
+ + + {copy.readingFrom} {host ?? copy.connectedComputer} + + + + {copy.destination} {owner.targetProfile ?? owner.profile} + +
+
+
+ +
+ {!current ? ( + + ) : ( + <> +
+
+ +
+

{current.title}

+

+ {current.label} · {current.turn_count} {copy.messages} +

+
+
+ {preview.isPending && } + {preview.isError && ( + + + + )} + {preview.data && ( +
+ {preview.data.truncated && ( +

{copy.previewLimit}

+ )} + {preview.data.messages.map((message, index) => ( +
+

+ {message.role === 'user' ? copy.you : current.label} +

+
+ +
+
+ ))} +
+ )} +
+
+

+ {preview.data?.already_imported ? copy.snapshot : copy.copyNotice} +

+ + {error && ( +
+ {copy.importError} {error} +
+ )} +
+ + )} +
+
+
+
+ ) +} diff --git a/apps/desktop/src/components/assistant-ui/markdown-text.tsx b/apps/desktop/src/components/assistant-ui/markdown-text.tsx index f0671fb1a8..23031e6514 100644 --- a/apps/desktop/src/components/assistant-ui/markdown-text.tsx +++ b/apps/desktop/src/components/assistant-ui/markdown-text.tsx @@ -449,6 +449,8 @@ interface MarkdownTextSurfaceProps { /** Disable artifact-card promotion for fenced blocks (reasoning text — a * model's scratchpad draft must not register artifact versions). */ disableArtifacts?: boolean + /** Foreign history must not load images or mount live transcript directives. */ + previewOnly?: boolean } // Headings shrink to chat scale rather than the prose default (h1≈xl). Kept @@ -537,7 +539,8 @@ function MarkdownTextSurface({ containerClassName, containerProps, defer, - disableArtifacts + disableArtifacts, + previewOnly }: MarkdownTextSurfaceProps) { const { status, text } = useMessagePartText() const isStreaming = status.type === 'running' @@ -564,8 +567,9 @@ function MarkdownTextSurface({ h4: ({ className, ...props }: ComponentProps<'h4'>) => (

), - p: (props: ComponentProps<'p'>) => , - a: MarkdownLink, + p: (props: ComponentProps<'p'>) => + previewOnly ?

: , + a: previewOnly ? ({ children }: ComponentProps<'a'>) => {children} : MarkdownLink, // Inline code must not vote when an ancestor resolves `dir="auto"` // (HTML's algorithm skips descendants that carry their own dir), // mirroring the CSS isolate that already keeps it out of the @@ -624,13 +628,13 @@ function MarkdownTextSurface({ td: ({ className, ...props }: ComponentProps<'td'>) => (

), - img: MarkdownImage, + img: previewOnly ? ({ alt }: ComponentProps<'img'>) => {alt} : MarkdownImage, // ```mermaid / ```svg fences route to their lazy renderers; substantial // html/svg/code fences promote to an artifact card that opens in the // right rail; every other language falls back to the Shiki-highlighted // code block. SyntaxHighlighter: (props: SyntaxHighlighterProps) => { - const artifact = disableArtifacts ? null : detectArtifact(props.language, props.code) + const artifact = disableArtifacts || previewOnly ? null : detectArtifact(props.language, props.code) if (artifact) { return @@ -646,7 +650,7 @@ function MarkdownTextSurface({ ) } }) as StreamdownTextComponents, - [disableArtifacts, isStreaming] + [disableArtifacts, isStreaming, previewOnly] ) if (text.length > MAX_MARKDOWN_CHARS) { diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index c99c0fbdce..007e631917 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -1,6 +1,42 @@ import { defineLocale } from './define-locale' export const ar = defineLocale({ + sessionImport: { + refresh: 'تحديث الجلسات', + title: 'المتابعة من تطبيق آخر', + subtitle: 'انقل محادثة إلى Hermes وتابع من حيث توقفت.', + action: 'استيراد جلسة', + readingFrom: 'القراءة من', + connectedComputer: 'الكمبيوتر المتصل', + destination: 'الاستيراد إلى', + all: 'الكل', + search: 'البحث في الجلسات المحملة', + scanning: 'جارٍ البحث عن المحادثات', + retry: 'حاول مرة أخرى', + scanError: 'تعذر العثور على الجلسات', + scanHelp: 'تحقق من اتصال الخادم ثم أعد المحاولة. قد تحتاج الخوادم القديمة إلى تحديث.', + empty: 'لا توجد محادثات', + emptyHelp: 'ستظهر هنا جلسات Claude Code وCodex الموجودة على هذا الخادم.', + noMatches: 'لا توجد محادثات مطابقة', + searchHelp: 'جرّب عنوانًا أو مجلدًا آخر، أو حمّل المزيد من الجلسات.', + skipped: 'تم تجاوز بعض السجلات الفارغة أو غير المقروءة أو الكبيرة جدًا.', + more: 'تحميل المزيد من الجلسات', + messages: 'رسائل', + choose: 'محادثة تستحق المتابعة', + chooseHelp: 'اختر جلسة لقراءة سجلها قبل نقلها إلى Hermes.', + back: 'العودة إلى الجلسات', + previewLoading: 'جارٍ فتح المعاينة', + previewError: 'المعاينة غير متاحة', + previewHelp: 'ربما تم نقل الملف الأصلي أو تغييره. حدّث القائمة وحاول مرة أخرى.', + previewLimit: 'تم اختصار المعاينة لتسهيل القراءة. يتم استيراد المحادثة كاملة.', + you: 'أنت', + snapshot: 'هذه المحادثة موجودة بالفعل في Hermes. افتح نسختك الحالية للمتابعة.', + copyNotice: 'ينسخ نص المحادثة دون تغيير الملفات الأصلية. لا يشمل مخرجات الأدوات أو الاستدلال.', + importing: 'جارٍ الاستيراد…', + open: 'فتح في Hermes', + continue: 'المتابعة في Hermes', + importError: 'تعذر استيراد هذه المحادثة.' + }, sendDiagnostics: { title: 'إرسال التشخيصات إلى Nous', privacyNotice: diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 6275a42ff7..7a1eef6084 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -3,6 +3,43 @@ import { FIELD_DESCRIPTIONS, FIELD_LABELS } from '@/app/settings/constants' import type { Translations } from './types' export const en: Translations = { + sessionImport: { + refresh: 'Refresh sessions', + title: 'Continue from another app', + subtitle: 'Bring a conversation into Hermes and pick up where you left off.', + action: 'Import session', + readingFrom: 'Reading from', + connectedComputer: 'the connected computer', + destination: 'Import into', + all: 'All', + search: 'Search loaded sessions', + scanning: 'Finding conversations', + retry: 'Try again', + scanError: 'Could not find sessions', + scanHelp: 'Check your backend connection, then try again. Older backends may need an update.', + empty: 'No conversations found', + emptyHelp: 'Claude Code and Codex sessions on this backend will appear here.', + noMatches: 'No matching conversations', + searchHelp: 'Try another title or folder, or load more sessions.', + skipped: 'Some logs were empty, unreadable, or too large to preview.', + more: 'Load more sessions', + messages: 'messages', + choose: 'A conversation worth continuing', + chooseHelp: 'Choose a session to read its history before bringing it into Hermes.', + back: 'Back to sessions', + previewLoading: 'Opening preview', + previewError: 'Preview unavailable', + previewHelp: 'The source may have moved or changed. Refresh the list and try again.', + previewLimit: 'Preview shortened for readability. The complete conversation is imported.', + you: 'You', + snapshot: 'This conversation is already in Hermes. Open your existing copy to continue.', + copyNotice: + 'Copies conversation text. Source files stay unchanged. Tool output and reasoning are not carried over.', + importing: 'Importing…', + open: 'Open in Hermes', + continue: 'Continue in Hermes', + importError: 'Could not import this conversation.' + }, common: { apply: 'Apply', back: 'Back', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 36d3408d9c..e7f5c7cbf4 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -3,6 +3,42 @@ import { defineFieldCopy } from '@/app/settings/field-copy' import { defineLocale } from './define-locale' export const ja = defineLocale({ + sessionImport: { + refresh: '一覧を更新', + title: '別のアプリから続ける', + subtitle: '会話をHermesに取り込み、続きを始めましょう。', + action: 'セッションを取り込む', + readingFrom: '読み込み元', + connectedComputer: '接続先のコンピューター', + destination: '取り込み先', + all: 'すべて', + search: '読み込み済みのセッションを検索', + scanning: '会話を検索中', + retry: '再試行', + scanError: 'セッションを取得できません', + scanHelp: 'バックエンドの接続を確認して再試行してください。古いバックエンドは更新が必要な場合があります。', + empty: '会話が見つかりません', + emptyHelp: 'このバックエンドのClaude CodeとCodexのセッションがここに表示されます。', + noMatches: '一致する会話がありません', + searchHelp: '別のタイトルやフォルダーを検索するか、セッションを追加で読み込んでください。', + skipped: '空、読み込み不可、または大きすぎるログをスキップしました。', + more: 'さらに読み込む', + messages: 'メッセージ', + choose: '会話の続きを始めましょう', + chooseHelp: 'セッションを選び、取り込む前に履歴を確認できます。', + back: 'セッションに戻る', + previewLoading: 'プレビューを開いています', + previewError: 'プレビューできません', + previewHelp: '元のファイルが移動または変更された可能性があります。一覧を更新してください。', + previewLimit: '読みやすいようにプレビューを省略しています。取り込み時は会話全体をコピーします。', + you: 'あなた', + snapshot: 'この会話は取り込み済みです。既存のコピーを開いて続けられます。', + copyNotice: '会話のテキストをコピーします。元のファイルは変更されません。ツール出力と推論は含まれません。', + importing: '取り込み中…', + open: 'Hermesで開く', + continue: 'Hermesで続ける', + importError: '会話を取り込めませんでした。' + }, common: { apply: '適用', back: '戻る', diff --git a/apps/desktop/src/i18n/ru.ts b/apps/desktop/src/i18n/ru.ts index fbeb912a24..a82750482a 100644 --- a/apps/desktop/src/i18n/ru.ts +++ b/apps/desktop/src/i18n/ru.ts @@ -24,6 +24,43 @@ const RU_NOUN = (count: number | string, one: string, few: string, many: string) } export const ru = defineLocale({ + sessionImport: { + refresh: 'Обновить сессии', + title: 'Продолжить из другого приложения', + subtitle: 'Перенесите разговор в Hermes и продолжите с того места, где остановились.', + action: 'Импортировать сессию', + readingFrom: 'Читаем с', + connectedComputer: 'подключённого компьютера', + destination: 'Импорт в', + all: 'Все', + search: 'Поиск по загруженным сессиям', + scanning: 'Поиск разговоров', + retry: 'Повторить', + scanError: 'Не удалось найти сессии', + scanHelp: 'Проверьте подключение к серверу и повторите попытку. Старому серверу может требоваться обновление.', + empty: 'Разговоров пока нет', + emptyHelp: 'Здесь появятся сессии Claude Code и Codex с этого сервера.', + noMatches: 'Совпадений нет', + searchHelp: 'Попробуйте другой заголовок или папку либо загрузите ещё сессии.', + skipped: 'Некоторые журналы пусты, недоступны или слишком велики для просмотра.', + more: 'Загрузить ещё сессии', + messages: 'сообщений', + choose: 'Разговор, который стоит продолжить', + chooseHelp: 'Выберите сессию, чтобы прочитать историю перед импортом в Hermes.', + back: 'Назад к сессиям', + previewLoading: 'Открываем просмотр', + previewError: 'Просмотр недоступен', + previewHelp: 'Исходный файл мог переместиться или измениться. Обновите список и повторите попытку.', + previewLimit: 'Просмотр сокращён для удобства чтения. Импортируется весь разговор.', + you: 'Вы', + snapshot: 'Этот разговор уже есть в Hermes. Откройте существующую копию, чтобы продолжить.', + copyNotice: + 'Копируется текст разговора. Исходные файлы не меняются. Вывод инструментов и рассуждения не переносятся.', + importing: 'Импорт…', + open: 'Открыть в Hermes', + continue: 'Продолжить в Hermes', + importError: 'Не удалось импортировать разговор.' + }, common: { apply: 'Применить', back: 'Назад', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index bacb24a5c7..b8dcc14adf 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -51,6 +51,42 @@ interface AuxTaskCopy { } export interface Translations { + sessionImport: { + refresh: string + title: string + subtitle: string + action: string + readingFrom: string + connectedComputer: string + destination: string + all: string + search: string + scanning: string + retry: string + scanError: string + scanHelp: string + empty: string + emptyHelp: string + noMatches: string + searchHelp: string + skipped: string + more: string + messages: string + choose: string + chooseHelp: string + back: string + previewLoading: string + previewError: string + previewHelp: string + previewLimit: string + you: string + snapshot: string + copyNotice: string + importing: string + open: string + continue: string + importError: string + } common: { apply: string back: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 9d6c61ba71..50a5a8b689 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -3,6 +3,42 @@ import { defineFieldCopy } from '@/app/settings/field-copy' import { defineLocale } from './define-locale' export const zhHant = defineLocale({ + sessionImport: { + refresh: '重新整理工作階段', + title: '從其他應用程式繼續', + subtitle: '將對話匯入 Hermes,接著上次的進度繼續。', + action: '匯入工作階段', + readingFrom: '讀取自', + connectedComputer: '已連線的電腦', + destination: '匯入至', + all: '全部', + search: '搜尋已載入的工作階段', + scanning: '正在尋找對話', + retry: '重試', + scanError: '無法尋找工作階段', + scanHelp: '請檢查後端連線並重試。舊版後端可能需要更新。', + empty: '找不到對話', + emptyHelp: '此後端上的 Claude Code 和 Codex 工作階段將顯示在這裡。', + noMatches: '沒有符合的對話', + searchHelp: '嘗試其他標題或資料夾,或載入更多工作階段。', + skipped: '部分記錄為空白、無法讀取或過大,已略過。', + more: '載入更多工作階段', + messages: '則訊息', + choose: '繼續一段對話', + chooseHelp: '選擇工作階段,在匯入 Hermes 前查看歷程記錄。', + back: '返回工作階段', + previewLoading: '正在開啟預覽', + previewError: '無法預覽', + previewHelp: '來源檔案可能已移動或變更。請重新整理清單後重試。', + previewLimit: '預覽已縮短,方便閱讀。匯入時會複製完整對話。', + you: '你', + snapshot: '此對話已匯入 Hermes。開啟現有副本即可繼續。', + copyNotice: '複製對話文字,不變更來源檔案。不包含工具輸出和推理內容。', + importing: '正在匯入…', + open: '在 Hermes 中開啟', + continue: '在 Hermes 中繼續', + importError: '無法匯入此對話。' + }, common: { apply: '套用', back: '返回', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 1c94cd2019..9ca3534e73 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -3,6 +3,42 @@ import { defineFieldCopy } from '@/app/settings/field-copy' import type { Translations } from './types' export const zh: Translations = { + sessionImport: { + refresh: '刷新会话', + title: '从其他应用继续', + subtitle: '将对话导入 Hermes,接着上次的进度继续。', + action: '导入会话', + readingFrom: '读取自', + connectedComputer: '已连接的计算机', + destination: '导入到', + all: '全部', + search: '搜索已加载的会话', + scanning: '正在查找对话', + retry: '重试', + scanError: '无法查找会话', + scanHelp: '请检查后端连接并重试。旧版后端可能需要更新。', + empty: '未找到对话', + emptyHelp: '此后端上的 Claude Code 和 Codex 会话将显示在这里。', + noMatches: '没有匹配的对话', + searchHelp: '尝试其他标题或文件夹,或加载更多会话。', + skipped: '部分日志为空、无法读取或过大,已跳过。', + more: '加载更多会话', + messages: '条消息', + choose: '继续一段对话', + chooseHelp: '选择会话,在导入 Hermes 前查看历史记录。', + back: '返回会话列表', + previewLoading: '正在打开预览', + previewError: '预览不可用', + previewHelp: '源文件可能已移动或更改。请刷新列表后重试。', + previewLimit: '预览已缩短,便于阅读。导入时会复制完整对话。', + you: '你', + snapshot: '此对话已导入 Hermes。打开现有副本即可继续。', + copyNotice: '复制对话文本,不更改源文件。不包含工具输出和推理内容。', + importing: '正在导入…', + open: '在 Hermes 中打开', + continue: '在 Hermes 中继续', + importError: '无法导入此对话。' + }, common: { apply: '应用', back: '返回', diff --git a/hermes_cli/foreign_session_browser.py b/hermes_cli/foreign_session_browser.py new file mode 100644 index 0000000000..cc194c6098 --- /dev/null +++ b/hermes_cli/foreign_session_browser.py @@ -0,0 +1,111 @@ +"""Backend-local discovery and previews for the desktop session importer.""" + +import hashlib +import re +import socket +from pathlib import Path + +from hermes_cli.foreign_sessions import _SOURCES, _SOURCE_LABELS, _SOURCE_DB_NAMES + +MAX_LOG_BYTES = 32 * 1024 * 1024 + + +def _display_title(parsed, source): + # Codex attachment envelopes put the actual request after a file listing. + # Keep the history intact; only the browser's display title skips that header. + first_user = next((turn["content"] for turn in parsed["turns"] if turn["role"] == "user"), "") + request = re.split(r"(?im)^#{1,6}\s*My request:\s*$", first_user, maxsplit=1) + title = request[-1].strip().splitlines()[0] if len(request) > 1 and request[-1].strip() else parsed["title_guess"] + return (title or _SOURCE_LABELS[source]).lstrip("# ")[:180] + + +def _candidates(source=None): + if source is not None and source not in _SOURCES: + raise ValueError("Unknown session source") + rows = [] + for name, (parts, pattern, recursive, _) in _SOURCES.items(): + if source and source != name: + continue + root = Path.home().joinpath(*parts).resolve() + if not root.is_dir(): + continue + for path in root.rglob(pattern) if recursive else root.glob(pattern): + resolved = path.resolve() + if not resolved.is_relative_to(root): + continue + try: + stat = resolved.stat() + except FileNotFoundError: + continue # A running source may rotate a log during discovery. + if not resolved.is_file(): + continue + handle = hashlib.sha256(f"{name}:{resolved}".encode()).hexdigest() + rows.append((stat.st_mtime, handle, name, resolved, stat.st_size)) + return sorted(rows, key=lambda row: (row[0], row[1]), reverse=True) + + +def _parse(candidate): + _, _, source, path, size = candidate + if size > MAX_LOG_BYTES: + raise ValueError("This log exceeds the 32 MB preview and import limit") + parsed = _SOURCES[source][3](path) + if not parsed["turns"]: + raise ValueError("This session has no readable conversation messages") + return parsed + + +def resolve_foreign_session(handle): + if not isinstance(handle, str) or len(handle) != 64: + raise ValueError("Unknown session. Refresh the list and try again") + for candidate in _candidates(): + if candidate[1] == handle: + return candidate, _parse(candidate) + raise ValueError("Session no longer available. Refresh the list and try again") + + +def list_foreign_sessions(source=None, offset=0, limit=25): + if not isinstance(offset, int) or offset < 0 or not isinstance(limit, int) or not 1 <= limit <= 50: + raise ValueError("Invalid session page") + candidates = _candidates(source) + rows, unreadable = [], 0 + # Page candidates before parsing. Empty or oversized logs cannot turn a + # request for 25 rows into an unbounded transcript scan. + for candidate in candidates[offset:offset + limit]: + mtime, handle, name, _, _ = candidate + try: + parsed = _parse(candidate) + except (ValueError, OSError): + unreadable += 1 + continue + rows.append({"id": handle, "source": name, "label": _SOURCE_LABELS[name], + "title": _display_title(parsed, name), + "cwd": parsed["cwd"], "mtime": mtime, "turn_count": len(parsed["turns"]), + "excerpt": parsed["turns"][0]["content"][:200]}) + next_offset = offset + limit + return {"sessions": rows, "next_offset": next_offset if next_offset < len(candidates) else None, + "host": socket.gethostname(), "unreadable": unreadable} + + +def foreign_origin(candidate, parsed): + return {"tool": _SOURCE_DB_NAMES[candidate[2]], "path": str(candidate[3]), + "foreign_session_id": parsed["session_id"]} + + +def preview_foreign_session(handle, db): + candidate, parsed = resolve_foreign_session(handle) + origin = foreign_origin(candidate, parsed) + existing = db.find_foreign_import(origin) + # A bounded preview avoids mounting thousands of messages in the renderer. + messages = [{**turn, "content": turn["content"][:8000]} for turn in parsed["turns"][-40:]] + return {"messages": messages, "total": len(parsed["turns"]), + "truncated": len(parsed["turns"]) > 40 or any(len(turn["content"]) > 8000 for turn in parsed["turns"][-40:]), + "already_imported": existing, "cwd": parsed["cwd"]} + + +def import_browser_session(handle, db, profile): + candidate, parsed = resolve_foreign_session(handle) + origin = foreign_origin(candidate, parsed) + cwd = parsed["cwd"] + return db.import_foreign_history(origin, parsed["turns"], + title=_display_title(parsed, candidate[2]), + cwd=cwd if cwd and Path(cwd).is_dir() else None, profile=profile) diff --git a/hermes_state_portability.py b/hermes_state_portability.py index d2da283c50..3858ccd9f5 100644 --- a/hermes_state_portability.py +++ b/hermes_state_portability.py @@ -80,6 +80,55 @@ _PROMPT_RESOLVED_SQL = "COALESCE(sp.prompt, s.system_prompt) AS _system_prompt_r class SessionPortabilityMixin: """See module docstring — mixin for SessionDB (Port cluster).""" + @staticmethod + def _find_foreign_import_on_conn(conn, origin): + rows = conn.execute("SELECT id, origin_json FROM sessions WHERE source = ? AND origin_json IS NOT NULL", + (origin["tool"],)).fetchall() + for row in rows: + imported = (safe_json_loads(row["origin_json"], default={}) or {}).get("imported_from", {}) + if imported.get("tool") != origin["tool"]: + continue + foreign_id = origin.get("foreign_session_id") + if ((foreign_id and foreign_id == imported.get("foreign_session_id")) + or (not foreign_id and imported.get("path") == origin["path"])): + return row["id"] + return None + + def find_foreign_import(self, origin): + with self._read_ctx() as conn: + return self._find_foreign_import_on_conn(conn, origin) + + def import_foreign_history(self, origin, messages, *, title, cwd, profile): + """Adopt or mint a foreign snapshot in one transaction, including its provenance. + + BEGIN IMMEDIATE serializes duplicate clicks across connections/processes. + Reuse the portability validator and message writer so counters and FTS + obey the same contract as ordinary transcript imports. + """ + import uuid + session_id = f"{time.strftime('%Y%m%d_%H%M%S')}_{uuid.uuid4().hex[:12]}" + normalized, errors = self._validate_import_payload([ + {"id": session_id, "source": origin["tool"], "title": title, + "cwd": cwd, "messages": messages}]) + if errors: + raise ValueError(errors[0]["error"]) + + def _do(conn): + existing = self._find_foreign_import_on_conn(conn, origin) + if existing: + return {"session_id": existing, "already_imported": True} + # Titles are globally unique within a profile. Preserve a readable + # title while giving unrelated conversations with the same text room. + item = normalized[0] + if conn.execute("SELECT 1 FROM sessions WHERE title = ?", (title,)).fetchone(): + item["session"]["title"] = f"{title} ({session_id[-12:]})" + self._import_session_row(conn, item["session"], item["messages"], session_id) + conn.execute("UPDATE sessions SET origin_json = ?, profile_name = ? WHERE id = ?", + (json.dumps({"imported_from": origin}), profile, session_id)) + return {"session_id": session_id, "already_imported": False} + + return self._execute_write(_do) + @classmethod def _compact_session_cols(cls) -> str: """``s.``-prefixed SELECT list of every SCHEMA_SQL ``sessions`` column except diff --git a/tests/tui_gateway/test_session_foreign.py b/tests/tui_gateway/test_session_foreign.py new file mode 100644 index 0000000000..fa8b4b93df --- /dev/null +++ b/tests/tui_gateway/test_session_foreign.py @@ -0,0 +1,97 @@ +"""Real foreign logs through the registered desktop RPCs and profile stores.""" + +import json +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +import pytest + + +def test_foreign_rpc_preview_import_and_profile_isolation(tmp_path, monkeypatch): + from hermes_state import SessionDB + from tui_gateway import server + + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes")) + monkeypatch.delenv("HERMES_DESKTOP", raising=False) + folder = tmp_path / ".claude" / "projects" / "project" + folder.mkdir(parents=True) + log = folder / "session.jsonl" + lines = [{"type": role, "sessionId": "foreign-one", "cwd": str(tmp_path), + "message": {"role": role, "content": content}} + for role, content in [("user", "# Files mentioned by the user:\nplan.md\n\n## My request:\nFix the import"), ("assistant", "Here is the fix."), + ("assistant", "And its test."), ("user", "Continue")]] + log.write_text("\n".join(map(json.dumps, lines)), encoding="utf-8") + original = log.read_bytes() + db = SessionDB(tmp_path / ".hermes" / "state.db") + monkeypatch.setattr(server, "_get_db", lambda: db) + monkeypatch.setattr(server, "_profile_home", lambda profile: None) + + def rpc(method, **params): + result = server._methods[f"session.foreign.{method}"](1, params) + assert "error" not in result, result + return result["result"] + + try: + page = rpc("list", limit=1) + handle = page["sessions"][0]["id"] + assert page["sessions"][0]["title"] == "Fix the import" + assert "path" not in page["sessions"][0] + preview = rpc("preview", id=handle) + assert preview["already_imported"] is None + assert db.session_count() == 0 + with ThreadPoolExecutor(max_workers=2) as executor: + results = list(executor.map(lambda _: rpc("import", id=handle), range(2))) + assert results[0]["session_id"] == results[1]["session_id"] + sid = results[0]["session_id"] + stored = db.get_session(sid) + assert json.loads(stored["origin_json"])["imported_from"]["foreign_session_id"] == "foreign-one" + history = db.get_messages(sid) + assert [message["role"] for message in history] == ["user", "assistant", "user"] + assert len(history) == preview["total"] == stored["message_count"] + assert rpc("preview", id=handle)["already_imported"] == sid + assert log.read_bytes() == original + with SessionDB(tmp_path / "other" / "state.db") as other: + monkeypatch.setattr(server, "_get_db", lambda: other) + assert rpc("preview", id=handle)["already_imported"] is None + assert rpc("import", id=handle)["session_id"] != sid + finally: + db.close() + + +def test_foreign_pages_confine_handles_and_failed_import_rolls_back(tmp_path, monkeypatch): + from hermes_cli import foreign_session_browser as browser + from hermes_state import SessionDB + + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes")) + folder = tmp_path / ".codex" / "sessions" + folder.mkdir(parents=True) + for index in range(3): + (folder / f"rollout-{index}.jsonl").write_text(json.dumps({ + "type": "response_item", "payload": {"type": "message", "role": "user", + "content": [{"type": "input_text", "text": f"Question {index}"}]}}), encoding="utf-8") + calls = [] + parse = browser._parse + monkeypatch.setattr(browser, "_parse", lambda row: (calls.append(row[1]), parse(row))[1]) + page = browser.list_foreign_sessions(limit=1) + assert len(calls) == 1 + next_page = browser.list_foreign_sessions(offset=page["next_offset"], limit=1) + assert page["sessions"][0]["id"] != next_page["sessions"][0]["id"] + with pytest.raises(ValueError): + browser.resolve_foreign_session(str(tmp_path / "secret.jsonl")) + # A fabricated well-shaped handle cannot become a file-read request. + with pytest.raises(ValueError): + browser.resolve_foreign_session("0" * 64) + db = SessionDB(tmp_path / ".hermes" / "state.db") + insert = db._insert_message_rows + def failing_insert(*args): + insert(*args) + raise RuntimeError("interrupted write") + monkeypatch.setattr(db, "_insert_message_rows", failing_insert) + try: + with pytest.raises(RuntimeError, match="interrupted write"): + browser.import_browser_session(page["sessions"][0]["id"], db, "default") + assert db.session_count() == 0 + finally: + db.close() diff --git a/tui_gateway/methods_session_foreign.py b/tui_gateway/methods_session_foreign.py new file mode 100644 index 0000000000..82fc7edf15 --- /dev/null +++ b/tui_gateway/methods_session_foreign.py @@ -0,0 +1,46 @@ +"""Desktop foreign-history browsing, scoped to the serving backend and profile.""" + +from .method_ctx import HandlerRegistry, bind_module + +_registry = HandlerRegistry() +method = _registry.method + + +@method("session.foreign.list") +def _foreign_list(rid, params): + from hermes_cli.foreign_session_browser import list_foreign_sessions + try: + return _ok(rid, list_foreign_sessions(params.get("source"), params.get("offset", 0), params.get("limit", 25))) + except ValueError as exc: + return _err(rid, -32602, str(exc)) + except OSError: + return _err(rid, -32000, "Could not read session folders on this backend") + + +def _foreign_history_request(rid, params, importing): + from hermes_cli.foreign_session_browser import import_browser_session, preview_foreign_session + try: + with _profile_db(params) as db: + if db is None: + return _db_unavailable_error(rid, code=-32000) + result = (import_browser_session(params.get("id"), db, _response_profile_name(params.get("profile"))) + if importing else preview_foreign_session(params.get("id"), db)) + return _ok(rid, result) + except ValueError as exc: + return _err(rid, -32602, str(exc)) + except OSError: + return _err(rid, -32000, "Could not read this session on the backend") + + +@method("session.foreign.preview") +def _foreign_preview(rid, params): + return _foreign_history_request(rid, params, False) + + +@method("session.foreign.import") +def _foreign_import(rid, params): + return _foreign_history_request(rid, params, True) + + +def register(server): + bind_module(globals(), server) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 671def454a..33ebf7ba25 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -158,6 +158,7 @@ _DETAIL_MODES = frozenset({"hidden", "collapsed", "expanded"}) # interrupts); voice.*/wake.* = SYNCHRONOUS faster-whisper install (300s); session.workspace.move = # git subprocess probes on an arbitrary (maybe slow) mount. _LONG_HANDLERS = frozenset({ + "session.foreign.list", "session.foreign.preview", "session.foreign.import", "billing.state", "subscription.state", "subscription.preview", "subscription.change", "subscription.resume", "subscription.upgrade", "usage.bars", "session.usage", "billing.step_up", "browser.manage", "cli.exec", "complete.path", "complete.slash", "llm.oneshot", "model.options", @@ -3206,7 +3207,7 @@ from . import ( # noqa: E402 methods_config_set as _methods_config_set, methods_images as _methods_images, methods_profiles as _methods_profiles, methods_prompt as _methods_prompt, methods_session as _methods_session, methods_tools as _methods_tools, prompt_turn as _prompt_turn, billing_view as _billing_view, - methods_projects as _methods_projects) + methods_projects as _methods_projects, methods_session_foreign as _methods_session_foreign) for _m in ( _session_reaper, _session_lifecycle, _session_workdir, _compute_host_bridge, _model_switch, @@ -3215,6 +3216,6 @@ for _m in ( _methods_complete_helpers, _methods_slash, _methods_voice, _methods_browser, _methods_browser_control, _methods_session, _methods_prompt, _methods_config, _methods_config_set, _methods_complete, _methods_tools, _methods_profiles, _methods_images, - _methods_bot_relay, _prompt_turn, _billing_view, _methods_projects): + _methods_bot_relay, _prompt_turn, _billing_view, _methods_projects, _methods_session_foreign): _m.register(sys.modules[__name__]) del _m From dfd0660fd7def84c8179d0e69b5304865ab79b7b Mon Sep 17 00:00:00 2001 From: Adolanium <94890352+Adolanium@users.noreply.github.com> Date: Sat, 5 Sep 2026 21:22:23 +0300 Subject: [PATCH 173/276] fix(desktop): skip inaccessible logs during foreign session discovery One unreadable or rotated log must not hide the rest of the list; discovery tolerates OSError from resolve()/stat() per entry and reuses the stat result. --- hermes_cli/foreign_session_browser.py | 13 ++++---- .../test_foreign_session_browser.py | 32 +++++++++++++++++++ 2 files changed, 39 insertions(+), 6 deletions(-) create mode 100644 tests/hermes_cli/test_foreign_session_browser.py diff --git a/hermes_cli/foreign_session_browser.py b/hermes_cli/foreign_session_browser.py index cc194c6098..627917e7e9 100644 --- a/hermes_cli/foreign_session_browser.py +++ b/hermes_cli/foreign_session_browser.py @@ -4,6 +4,7 @@ import hashlib import re import socket from pathlib import Path +from stat import S_ISREG from hermes_cli.foreign_sessions import _SOURCES, _SOURCE_LABELS, _SOURCE_DB_NAMES @@ -30,14 +31,14 @@ def _candidates(source=None): if not root.is_dir(): continue for path in root.rglob(pattern) if recursive else root.glob(pattern): - resolved = path.resolve() - if not resolved.is_relative_to(root): - continue try: + resolved = path.resolve() + if not resolved.is_relative_to(root): + continue stat = resolved.stat() - except FileNotFoundError: - continue # A running source may rotate a log during discovery. - if not resolved.is_file(): + except OSError: + continue # One inaccessible or rotated log must not hide the rest. + if not S_ISREG(stat.st_mode): continue handle = hashlib.sha256(f"{name}:{resolved}".encode()).hexdigest() rows.append((stat.st_mtime, handle, name, resolved, stat.st_size)) diff --git a/tests/hermes_cli/test_foreign_session_browser.py b/tests/hermes_cli/test_foreign_session_browser.py new file mode 100644 index 0000000000..9f2565799e --- /dev/null +++ b/tests/hermes_cli/test_foreign_session_browser.py @@ -0,0 +1,32 @@ +"""Discovery keeps readable sessions available when a neighboring log is inaccessible.""" + +import json +from pathlib import Path + +import pytest + +from hermes_cli.foreign_session_browser import list_foreign_sessions + + +@pytest.mark.parametrize("operation", ["resolve", "stat"]) +def test_discovery_skips_inaccessible_log(tmp_path, monkeypatch, operation): + monkeypatch.setattr(Path, "home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes")) + folder = tmp_path / ".codex" / "sessions" + folder.mkdir(parents=True) + for name in ("readable", "inaccessible"): + (folder / f"rollout-{name}.jsonl").write_text(json.dumps({ + "type": "response_item", "payload": {"type": "message", "role": "user", + "content": [{"type": "input_text", "text": name}]}, + }), encoding="utf-8") + + original = getattr(Path, operation) + + def access(path, *args, **kwargs): + if path.name == "rollout-inaccessible.jsonl": + raise PermissionError("Log access denied") + return original(path, *args, **kwargs) + + monkeypatch.setattr(Path, operation, access) + page = list_foreign_sessions("codex") + assert [session["title"] for session in page["sessions"]] == ["readable"] From fe49acb67099341b1e6ae4bd076982b231d30eda Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:55:29 -0700 Subject: [PATCH 174/276] refactor(desktop): import-session entry is a sidebar nav row; browser module named as a foreign_sessions sibling - Sidebar: the entry joins SIDEBAR_NAV (same chrome, active state, data-tour handle as the other rows) instead of a one-off Button below the rail; i18n moves to sidebar.nav['session-import']. - Reuse common.retry / common.refresh / common.back instead of duplicating them under sessionImport in six locales. - hermes_cli/foreign_session_browser.py -> foreign_sessions_browser.py so it sorts as a sibling of the foreign_sessions module it extends. --- apps/desktop/src/app/chat/sidebar/index.tsx | 19 +++++++++---------- apps/desktop/src/app/session-import/index.tsx | 8 ++++---- apps/desktop/src/app/types.ts | 3 ++- apps/desktop/src/i18n/ar.ts | 6 ++---- apps/desktop/src/i18n/en.ts | 6 ++---- apps/desktop/src/i18n/ja.ts | 6 ++---- apps/desktop/src/i18n/ru.ts | 6 ++---- apps/desktop/src/i18n/types.ts | 3 --- apps/desktop/src/i18n/zh-hant.ts | 6 ++---- apps/desktop/src/i18n/zh.ts | 6 ++---- ...browser.py => foreign_sessions_browser.py} | 0 ...er.py => test_foreign_sessions_browser.py} | 2 +- tests/tui_gateway/test_session_foreign.py | 2 +- tui_gateway/methods_session_foreign.py | 4 ++-- 14 files changed, 31 insertions(+), 46 deletions(-) rename hermes_cli/{foreign_session_browser.py => foreign_sessions_browser.py} (100%) rename tests/hermes_cli/{test_foreign_session_browser.py => test_foreign_sessions_browser.py} (94%) diff --git a/apps/desktop/src/app/chat/sidebar/index.tsx b/apps/desktop/src/app/chat/sidebar/index.tsx index 0aefa90e30..6fcd656da3 100644 --- a/apps/desktop/src/app/chat/sidebar/index.tsx +++ b/apps/desktop/src/app/chat/sidebar/index.tsx @@ -3,10 +3,9 @@ import { sortableKeyboardCoordinates } from '@dnd-kit/sortable' import { useStore } from '@nanostores/react' import type * as React from 'react' import { useCallback, useEffect, useMemo, useRef, useState } from 'react' -import { useLocation, useNavigate } from 'react-router' +import { useLocation } from 'react-router' import { PlatformAvatar } from '@/app/messaging/platform-icon' -import { SESSION_IMPORT_ROUTE } from '@/app/routes' import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' import { ContextMenu, ContextMenuContent, ContextMenuTrigger } from '@/components/ui/context-menu' @@ -140,6 +139,7 @@ import { ARTIFACTS_ROUTE, CRON_ROUTE, MESSAGING_ROUTE, + SESSION_IMPORT_ROUTE, SIDEBAR_NAV_AREA, type SidebarNavContribution, SKILLS_ROUTE @@ -226,6 +226,12 @@ const SIDEBAR_NAV: SidebarNavItem[] = [ icon: props => , route: CRON_ROUTE, keybindActionId: 'nav.cron' + }, + { + id: 'session-import', + label: '', + icon: props => , + route: SESSION_IMPORT_ROUTE } ] @@ -334,7 +340,6 @@ export function ChatSidebar({ const { t } = useI18n() const s = t.sidebar const { pathname } = useLocation() - const navigate = useNavigate() // Contributed nav rows (plugins pairing a page with a sidebar entry) render // below the built-ins with the same chrome; active = at their route. const navContributions = useContributions(SIDEBAR_NAV_AREA) @@ -1499,6 +1504,7 @@ export function ChatSidebar({ (item.id === 'messaging' && currentView === 'messaging') || (item.id === 'artifacts' && currentView === 'artifacts') || (item.id === 'cron' && currentView === 'cron') || + (item.id === 'session-import' && currentView === 'session-import') || // Contributed rows light up at their own route. (Boolean(item.route) && pathname === item.route) @@ -1619,13 +1625,6 @@ export function ChatSidebar({ -
- -
- {showSessionSections && (
{ void sessions.refetch() @@ -183,7 +183,7 @@ export function SessionImportView({ owner, onClose, onOpenSession }: SessionImpo {sessions.isError && ( )} @@ -261,7 +261,7 @@ export function SessionImportView({ owner, onClose, onOpenSession }: SessionImpo

{current.title}

@@ -274,7 +274,7 @@ export function SessionImportView({ owner, onClose, onOpenSession }: SessionImpo {preview.isError && ( )} diff --git a/apps/desktop/src/app/types.ts b/apps/desktop/src/app/types.ts index 57e85a0a9c..4d3f7545f1 100644 --- a/apps/desktop/src/app/types.ts +++ b/apps/desktop/src/app/types.ts @@ -162,7 +162,8 @@ export type CommandDispatchResponse = | SendCommandDispatchResponse | PrefillCommandDispatchResponse -export type SidebarNavId = 'artifacts' | 'command-center' | 'cron' | 'messaging' | 'new-session' | 'settings' | 'skills' +export type SidebarNavId = + 'artifacts' | 'command-center' | 'cron' | 'messaging' | 'new-session' | 'session-import' | 'settings' | 'skills' export interface SidebarNavItem { /** Built-in view id, or a contributed row's namespaced contribution id. */ diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 007e631917..abc18ada1b 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -2,7 +2,6 @@ import { defineLocale } from './define-locale' export const ar = defineLocale({ sessionImport: { - refresh: 'تحديث الجلسات', title: 'المتابعة من تطبيق آخر', subtitle: 'انقل محادثة إلى Hermes وتابع من حيث توقفت.', action: 'استيراد جلسة', @@ -12,7 +11,6 @@ export const ar = defineLocale({ all: 'الكل', search: 'البحث في الجلسات المحملة', scanning: 'جارٍ البحث عن المحادثات', - retry: 'حاول مرة أخرى', scanError: 'تعذر العثور على الجلسات', scanHelp: 'تحقق من اتصال الخادم ثم أعد المحاولة. قد تحتاج الخوادم القديمة إلى تحديث.', empty: 'لا توجد محادثات', @@ -24,7 +22,6 @@ export const ar = defineLocale({ messages: 'رسائل', choose: 'محادثة تستحق المتابعة', chooseHelp: 'اختر جلسة لقراءة سجلها قبل نقلها إلى Hermes.', - back: 'العودة إلى الجلسات', previewLoading: 'جارٍ فتح المعاينة', previewError: 'المعاينة غير متاحة', previewHelp: 'ربما تم نقل الملف الأصلي أو تغييره. حدّث القائمة وحاول مرة أخرى.', @@ -1734,7 +1731,8 @@ export const ar = defineLocale({ chat: 'المحادثة', settings: 'الإعدادات', cron: 'المهام المجدولة', - agents: 'الوكلاء' + agents: 'الوكلاء', + 'session-import': 'استيراد جلسة' }, searchAria: 'البحث في الجلسات', searchPlaceholder: 'البحث في الجلسات...', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 7a1eef6084..a3d966fb3f 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -4,7 +4,6 @@ import type { Translations } from './types' export const en: Translations = { sessionImport: { - refresh: 'Refresh sessions', title: 'Continue from another app', subtitle: 'Bring a conversation into Hermes and pick up where you left off.', action: 'Import session', @@ -14,7 +13,6 @@ export const en: Translations = { all: 'All', search: 'Search loaded sessions', scanning: 'Finding conversations', - retry: 'Try again', scanError: 'Could not find sessions', scanHelp: 'Check your backend connection, then try again. Older backends may need an update.', empty: 'No conversations found', @@ -26,7 +24,6 @@ export const en: Translations = { messages: 'messages', choose: 'A conversation worth continuing', chooseHelp: 'Choose a session to read its history before bringing it into Hermes.', - back: 'Back to sessions', previewLoading: 'Opening preview', previewError: 'Preview unavailable', previewHelp: 'The source may have moved or changed. Refresh the list and try again.', @@ -2360,7 +2357,8 @@ export const en: Translations = { skills: 'Capabilities', messaging: 'Messaging', artifacts: 'Artifacts', - cron: 'Scheduled jobs' + cron: 'Scheduled jobs', + 'session-import': 'Import session' }, searchAria: 'Search sessions', searchPlaceholder: 'Search sessions…', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index e7f5c7cbf4..f49de611dc 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -4,7 +4,6 @@ import { defineLocale } from './define-locale' export const ja = defineLocale({ sessionImport: { - refresh: '一覧を更新', title: '別のアプリから続ける', subtitle: '会話をHermesに取り込み、続きを始めましょう。', action: 'セッションを取り込む', @@ -14,7 +13,6 @@ export const ja = defineLocale({ all: 'すべて', search: '読み込み済みのセッションを検索', scanning: '会話を検索中', - retry: '再試行', scanError: 'セッションを取得できません', scanHelp: 'バックエンドの接続を確認して再試行してください。古いバックエンドは更新が必要な場合があります。', empty: '会話が見つかりません', @@ -26,7 +24,6 @@ export const ja = defineLocale({ messages: 'メッセージ', choose: '会話の続きを始めましょう', chooseHelp: 'セッションを選び、取り込む前に履歴を確認できます。', - back: 'セッションに戻る', previewLoading: 'プレビューを開いています', previewError: 'プレビューできません', previewHelp: '元のファイルが移動または変更された可能性があります。一覧を更新してください。', @@ -2034,7 +2031,8 @@ export const ja = defineLocale({ skills: 'スキルとツール', messaging: 'メッセージング', artifacts: 'アーティファクト', - cron: 'スケジュール済みジョブ' + cron: 'スケジュール済みジョブ', + 'session-import': 'セッションを取り込む' }, searchAria: 'セッションを検索', searchPlaceholder: 'セッションを検索…', diff --git a/apps/desktop/src/i18n/ru.ts b/apps/desktop/src/i18n/ru.ts index a82750482a..a43567045f 100644 --- a/apps/desktop/src/i18n/ru.ts +++ b/apps/desktop/src/i18n/ru.ts @@ -25,7 +25,6 @@ const RU_NOUN = (count: number | string, one: string, few: string, many: string) export const ru = defineLocale({ sessionImport: { - refresh: 'Обновить сессии', title: 'Продолжить из другого приложения', subtitle: 'Перенесите разговор в Hermes и продолжите с того места, где остановились.', action: 'Импортировать сессию', @@ -35,7 +34,6 @@ export const ru = defineLocale({ all: 'Все', search: 'Поиск по загруженным сессиям', scanning: 'Поиск разговоров', - retry: 'Повторить', scanError: 'Не удалось найти сессии', scanHelp: 'Проверьте подключение к серверу и повторите попытку. Старому серверу может требоваться обновление.', empty: 'Разговоров пока нет', @@ -47,7 +45,6 @@ export const ru = defineLocale({ messages: 'сообщений', choose: 'Разговор, который стоит продолжить', chooseHelp: 'Выберите сессию, чтобы прочитать историю перед импортом в Hermes.', - back: 'Назад к сессиям', previewLoading: 'Открываем просмотр', previewError: 'Просмотр недоступен', previewHelp: 'Исходный файл мог переместиться или измениться. Обновите список и повторите попытку.', @@ -2396,7 +2393,8 @@ export const ru = defineLocale({ skills: 'Возможности', messaging: 'Сообщения', artifacts: 'Артефакты', - cron: 'Запланированные задачи' + cron: 'Запланированные задачи', + 'session-import': 'Импортировать сессию' }, searchAria: 'Поиск сеансов', searchPlaceholder: 'Поиск сеансов…', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index b8dcc14adf..40557297a9 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -52,7 +52,6 @@ interface AuxTaskCopy { export interface Translations { sessionImport: { - refresh: string title: string subtitle: string action: string @@ -62,7 +61,6 @@ export interface Translations { all: string search: string scanning: string - retry: string scanError: string scanHelp: string empty: string @@ -74,7 +72,6 @@ export interface Translations { messages: string choose: string chooseHelp: string - back: string previewLoading: string previewError: string previewHelp: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 50a5a8b689..f73e8883ef 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -4,7 +4,6 @@ import { defineLocale } from './define-locale' export const zhHant = defineLocale({ sessionImport: { - refresh: '重新整理工作階段', title: '從其他應用程式繼續', subtitle: '將對話匯入 Hermes,接著上次的進度繼續。', action: '匯入工作階段', @@ -14,7 +13,6 @@ export const zhHant = defineLocale({ all: '全部', search: '搜尋已載入的工作階段', scanning: '正在尋找對話', - retry: '重試', scanError: '無法尋找工作階段', scanHelp: '請檢查後端連線並重試。舊版後端可能需要更新。', empty: '找不到對話', @@ -26,7 +24,6 @@ export const zhHant = defineLocale({ messages: '則訊息', choose: '繼續一段對話', chooseHelp: '選擇工作階段,在匯入 Hermes 前查看歷程記錄。', - back: '返回工作階段', previewLoading: '正在開啟預覽', previewError: '無法預覽', previewHelp: '來源檔案可能已移動或變更。請重新整理清單後重試。', @@ -1959,7 +1956,8 @@ export const zhHant = defineLocale({ skills: '技能與工具', messaging: '訊息平台', artifacts: '成品', - cron: '排程工作' + cron: '排程工作', + 'session-import': '匯入工作階段' }, searchAria: '搜尋工作階段', searchPlaceholder: '搜尋工作階段…', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 9ca3534e73..501e19b3a6 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -4,7 +4,6 @@ import type { Translations } from './types' export const zh: Translations = { sessionImport: { - refresh: '刷新会话', title: '从其他应用继续', subtitle: '将对话导入 Hermes,接着上次的进度继续。', action: '导入会话', @@ -14,7 +13,6 @@ export const zh: Translations = { all: '全部', search: '搜索已加载的会话', scanning: '正在查找对话', - retry: '重试', scanError: '无法查找会话', scanHelp: '请检查后端连接并重试。旧版后端可能需要更新。', empty: '未找到对话', @@ -26,7 +24,6 @@ export const zh: Translations = { messages: '条消息', choose: '继续一段对话', chooseHelp: '选择会话,在导入 Hermes 前查看历史记录。', - back: '返回会话列表', previewLoading: '正在打开预览', previewError: '预览不可用', previewHelp: '源文件可能已移动或更改。请刷新列表后重试。', @@ -2526,7 +2523,8 @@ export const zh: Translations = { skills: '技能与工具', messaging: '消息平台', artifacts: '产物', - cron: '定时任务' + cron: '定时任务', + 'session-import': '导入会话' }, searchAria: '搜索会话', searchPlaceholder: '搜索会话…', diff --git a/hermes_cli/foreign_session_browser.py b/hermes_cli/foreign_sessions_browser.py similarity index 100% rename from hermes_cli/foreign_session_browser.py rename to hermes_cli/foreign_sessions_browser.py diff --git a/tests/hermes_cli/test_foreign_session_browser.py b/tests/hermes_cli/test_foreign_sessions_browser.py similarity index 94% rename from tests/hermes_cli/test_foreign_session_browser.py rename to tests/hermes_cli/test_foreign_sessions_browser.py index 9f2565799e..e43ba6eb2f 100644 --- a/tests/hermes_cli/test_foreign_session_browser.py +++ b/tests/hermes_cli/test_foreign_sessions_browser.py @@ -5,7 +5,7 @@ from pathlib import Path import pytest -from hermes_cli.foreign_session_browser import list_foreign_sessions +from hermes_cli.foreign_sessions_browser import list_foreign_sessions @pytest.mark.parametrize("operation", ["resolve", "stat"]) diff --git a/tests/tui_gateway/test_session_foreign.py b/tests/tui_gateway/test_session_foreign.py index fa8b4b93df..355d2053c5 100644 --- a/tests/tui_gateway/test_session_foreign.py +++ b/tests/tui_gateway/test_session_foreign.py @@ -60,7 +60,7 @@ def test_foreign_rpc_preview_import_and_profile_isolation(tmp_path, monkeypatch) def test_foreign_pages_confine_handles_and_failed_import_rolls_back(tmp_path, monkeypatch): - from hermes_cli import foreign_session_browser as browser + from hermes_cli import foreign_sessions_browser as browser from hermes_state import SessionDB monkeypatch.setattr(Path, "home", lambda: tmp_path) diff --git a/tui_gateway/methods_session_foreign.py b/tui_gateway/methods_session_foreign.py index 82fc7edf15..ba37ad8580 100644 --- a/tui_gateway/methods_session_foreign.py +++ b/tui_gateway/methods_session_foreign.py @@ -8,7 +8,7 @@ method = _registry.method @method("session.foreign.list") def _foreign_list(rid, params): - from hermes_cli.foreign_session_browser import list_foreign_sessions + from hermes_cli.foreign_sessions_browser import list_foreign_sessions try: return _ok(rid, list_foreign_sessions(params.get("source"), params.get("offset", 0), params.get("limit", 25))) except ValueError as exc: @@ -18,7 +18,7 @@ def _foreign_list(rid, params): def _foreign_history_request(rid, params, importing): - from hermes_cli.foreign_session_browser import import_browser_session, preview_foreign_session + from hermes_cli.foreign_sessions_browser import import_browser_session, preview_foreign_session try: with _profile_db(params) as db: if db is None: From f3400ce7455b7f416cf197db8c46be375a1637cf Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:55:29 -0700 Subject: [PATCH 175/276] fix(cli): blank user turn no longer crashes foreign session discovery _first_user_line indexed splitlines()[0] on a whitespace-only user message (image-only / tool-only turn) and raised IndexError, which aborted the whole CLI picker and the desktop session.foreign.list RPC. partition("\n") yields "" for blank text and the loop moves on to the first real user line. Reported first in #92290. --- hermes_cli/foreign_sessions.py | 2 +- tests/hermes_cli/test_foreign_sessions.py | 17 +++++++++++++++++ 2 files changed, 18 insertions(+), 1 deletion(-) diff --git a/hermes_cli/foreign_sessions.py b/hermes_cli/foreign_sessions.py index e34fca7505..4c1983028c 100644 --- a/hermes_cli/foreign_sessions.py +++ b/hermes_cli/foreign_sessions.py @@ -108,7 +108,7 @@ def _message_turn(message: Any) -> Optional[Tuple[str, str]]: def _first_user_line(turns: List[Tuple[str, str]]) -> Optional[str]: for role, text in turns: - if role == "user" and (line := text.strip().splitlines()[0].strip()): + if role == "user" and (line := text.strip().partition("\n")[0].strip()): return line[:_TITLE_MAX * 2] return None diff --git a/tests/hermes_cli/test_foreign_sessions.py b/tests/hermes_cli/test_foreign_sessions.py index a522068971..5004776be1 100644 --- a/tests/hermes_cli/test_foreign_sessions.py +++ b/tests/hermes_cli/test_foreign_sessions.py @@ -258,3 +258,20 @@ def test_leading_assistant_gets_single_stub(tmp_path): _assert_alternating(parsed["turns"]) assert len(parsed["turns"]) == 2 assert parsed["turns"][0]["role"] == "user" + + +def test_whitespace_only_user_turn_does_not_break_discovery(tmp_path): + """A blank user message (image-only / tool-only turn) must not crash listing; the title + comes from the first non-blank user line and the blank turn is dropped.""" + project = tmp_path / ".claude" / "projects" / "p" + project.mkdir(parents=True) + f = project / "blank.jsonl" + lines = [ + {"type": "user", "sessionId": "w", "message": {"role": "user", "content": " \n "}}, + {"type": "assistant", "sessionId": "w", "message": {"role": "assistant", "content": "hi"}}, + {"type": "user", "sessionId": "w", "message": {"role": "user", "content": "real question"}}, + ] + f.write_text("\n".join(json.dumps(line) for line in lines) + "\n", encoding="utf-8") + listed = _list_sessions("claude", project.parent) + assert [s.title_guess for s in listed] == ["real question"] + assert listed[0].turn_count == 3 # leading assistant reply gets the user stub From 0168bb27ad6e307a41bf4dc42f24a90f62097fbe Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:56:54 -0700 Subject: [PATCH 176/276] refactor(cli): one foreign-log walker for the CLI picker and the desktop browser hermes_cli.foreign_sessions._walk owns discovery (glob per source, per-entry OSError tolerance, symlink-escape and regular-file checks, newest first); the CLI listing and the desktop browser's candidate rows both consume it instead of carrying two scans of the same directories. --- hermes_cli/foreign_sessions.py | 32 ++++++++++++++++++-------- hermes_cli/foreign_sessions_browser.py | 26 ++++----------------- 2 files changed, 28 insertions(+), 30 deletions(-) diff --git a/hermes_cli/foreign_sessions.py b/hermes_cli/foreign_sessions.py index 4c1983028c..9ace1d7d97 100644 --- a/hermes_cli/foreign_sessions.py +++ b/hermes_cli/foreign_sessions.py @@ -12,6 +12,7 @@ import uuid from dataclasses import dataclass from datetime import datetime from pathlib import Path +from stat import S_ISREG from typing import Any, Dict, List, Optional, Tuple # User-message texts that are really injected context wrappers, not typed input. @@ -169,20 +170,33 @@ _SOURCES = { } -def _list_sessions(source: str, root: Optional[Path]) -> List[ForeignSession]: - default_root, pattern, recursive, parse = _SOURCES[source] - root = Path(root) if root else Path.home().joinpath(*default_root) - results: List[ForeignSession] = [] - for jsonl in sorted((root.rglob(pattern) if recursive else root.glob(pattern)) if root.is_dir() else ()): +def _walk(source: str, root: Optional[Path] = None) -> List[Tuple[Path, os.stat_result]]: + """Regular log files of *source* under *root* (default ``~/``) as ``(path, stat)``, + newest first. Symlinks escaping the root and unreadable/rotated entries are skipped, so one + bad file never hides the rest. Shared by the CLI picker and the desktop browser.""" + default_root, pattern, recursive, _ = _SOURCES[source] + root = (Path(root) if root else Path.home().joinpath(*default_root)).resolve() + found: List[Tuple[Path, os.stat_result]] = [] + for path in (root.rglob(pattern) if recursive else root.glob(pattern)) if root.is_dir() else (): try: - mtime = jsonl.stat().st_mtime + resolved = path.resolve() + st = resolved.stat() except OSError: continue - parsed = parse(jsonl) + if resolved.is_relative_to(root) and S_ISREG(st.st_mode): + found.append((resolved, st)) + found.sort(key=lambda item: item[1].st_mtime, reverse=True) + return found + + +def _list_sessions(source: str, root: Optional[Path]) -> List[ForeignSession]: + parse = _SOURCES[source][3] + results: List[ForeignSession] = [] + for path, st in _walk(source, root): + parsed = parse(path) if parsed["turns"]: - results.append(ForeignSession(source, jsonl, mtime, parsed["cwd"], parsed["title_guess"], + results.append(ForeignSession(source, path, st.st_mtime, parsed["cwd"], parsed["title_guess"], len(parsed["turns"]), parsed["session_id"])) - results.sort(key=lambda s: s.mtime, reverse=True) return results diff --git a/hermes_cli/foreign_sessions_browser.py b/hermes_cli/foreign_sessions_browser.py index 627917e7e9..c49ac73865 100644 --- a/hermes_cli/foreign_sessions_browser.py +++ b/hermes_cli/foreign_sessions_browser.py @@ -4,9 +4,8 @@ import hashlib import re import socket from pathlib import Path -from stat import S_ISREG -from hermes_cli.foreign_sessions import _SOURCES, _SOURCE_LABELS, _SOURCE_DB_NAMES +from hermes_cli.foreign_sessions import _SOURCE_DB_NAMES, _SOURCE_LABELS, _SOURCES, _walk MAX_LOG_BYTES = 32 * 1024 * 1024 @@ -21,27 +20,12 @@ def _display_title(parsed, source): def _candidates(source=None): + """``(mtime, handle, source, path, size)`` rows across sources, newest first. The handle is + the only identifier handed to the client; a request can never name a path.""" if source is not None and source not in _SOURCES: raise ValueError("Unknown session source") - rows = [] - for name, (parts, pattern, recursive, _) in _SOURCES.items(): - if source and source != name: - continue - root = Path.home().joinpath(*parts).resolve() - if not root.is_dir(): - continue - for path in root.rglob(pattern) if recursive else root.glob(pattern): - try: - resolved = path.resolve() - if not resolved.is_relative_to(root): - continue - stat = resolved.stat() - except OSError: - continue # One inaccessible or rotated log must not hide the rest. - if not S_ISREG(stat.st_mode): - continue - handle = hashlib.sha256(f"{name}:{resolved}".encode()).hexdigest() - rows.append((stat.st_mtime, handle, name, resolved, stat.st_size)) + rows = [(st.st_mtime, hashlib.sha256(f"{name}:{path}".encode()).hexdigest(), name, path, st.st_size) + for name in _SOURCES if source in (None, name) for path, st in _walk(name)] return sorted(rows, key=lambda row: (row[0], row[1]), reverse=True) From fa869157cee3b07774541d846c41ddaaed98da46 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 03:57:57 -0700 Subject: [PATCH 177/276] docs(sessions): Desktop import-session entry next to the CLI importer --- website/docs/user-guide/sessions.md | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/website/docs/user-guide/sessions.md b/website/docs/user-guide/sessions.md index 67fa40e093..f90ea00f01 100644 --- a/website/docs/user-guide/sessions.md +++ b/website/docs/user-guide/sessions.md @@ -663,6 +663,14 @@ the id plus a ready-to-paste `hermes --resume ` command. `--resume @claude` / `--resume @codex` show the same picker and drop you straight into the imported conversation. +**Hermes Desktop** has the same importer under **Import session** in the +sidebar (also in the command palette). It lists the logs on the machine the +connected backend runs on — not the computer running the app — shows a +read-only preview, and **Continue in Hermes** copies the conversation into the +selected profile. Browsing never writes to your session store, importing never +touches the source file, and importing the same log twice opens the existing +copy instead of making another. + What carries over: the ordered user/assistant conversation, with tool activity condensed to short `[ran tool: …]` notes inside assistant turns. System prompts, injected context, reasoning traces, and raw tool output are From 45646f4d09bb2abf6c7d16569afba92de929dbda Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sun, 6 Sep 2026 09:11:04 -0700 Subject: [PATCH 178/276] feat(nous): anthropic_wire=auto decides a session's wire from its first response Portal will serve anthropic/* from more than one upstream (OpenRouter passthrough today; GMI/Vertex once it is back online). The native Messages wire is the better transport but is only safe where the upstream keeps prompt-cache routing sticky: measured false on the OpenRouter path (14-20% of consecutive calls re-write the previous turn; #104284 moved the default to chat), untested on GMI. Hermes cannot see the upstream in the request, only in the response: OpenRouter stamps `provider` (chat wire) and mints `gen--` ids; GMI/Vertex returns Anthropic-native `msg_...` ids and no provider. `auto` therefore starts every session on chat (correct on both upstreams), classifies the first response, and switches that session to native only when the upstream is GMI AND `agent/nous_wire.py::GMI_NATIVE_WIRE_CLEARED` is True. The switch is scheduled at response time and applied at the start of the next iteration (turn_iteration_prep), so nothing is rebuilt while a response is being consumed; it goes through switch_model so the client, cache policy and _primary_runtime stay consistent. One decision per session, call 1 only; unknown upstream never switches; a failed switch logs and stays. GMI_NATIVE_WIRE_CLEARED is False: until the 20x6 concurrency probe (evals/postmortem/live_ab) is clean on a GMI-served anthropic/* id on the native wire, `auto` behaves exactly like `chat`. Flipping it is the whole rollout once GMI is measured. Default stays `chat`. Tests (17): classifier on real Portal response shapes from both wires and both upstreams; chat for openrouter/unknown, GMI gated on the flag; one decision per session, call 1 only, explicit chat/native never auto-switch, other providers/models untouched, switch failure swallowed and final; record_response_usage on a real AIAgent invokes the hook once. Live (auto, real Portal, Fable 5.1): arm A, real classification (OpenRouter today) - stays on chat through a tool loop and a second turn, cache 97-99%. Arm B, classifier forced to gmi with the flag on - call 1 on chat, switch applied before call 2, calls 2-3 on the native wire in the same session, tool result and both turns correct, cache 97-99%. An earlier shape that switched inside the response path broke call 1 (SimpleNamespace has no .content); the scheduled apply is why. --- agent/nous_wire.py | 112 ++++++++++++++ agent/turn_iteration_prep.py | 6 + agent/turn_usage.py | 5 + hermes_cli/config_defaults.py | 10 +- hermes_cli/providers.py | 7 +- tests/agent/test_nous_wire_auto.py | 139 ++++++++++++++++++ website/docs/user-guide/configuring-models.md | 6 +- 7 files changed, 277 insertions(+), 8 deletions(-) create mode 100644 agent/nous_wire.py create mode 100644 tests/agent/test_nous_wire_auto.py diff --git a/agent/nous_wire.py b/agent/nous_wire.py new file mode 100644 index 0000000000..baf602abde --- /dev/null +++ b/agent/nous_wire.py @@ -0,0 +1,112 @@ +"""Nous Portal ``anthropic/*`` wire selection when ``nous.anthropic_wire`` is ``auto``. + +Portal serves Claude two ways and Hermes cannot tell which from the request: an OpenRouter +passthrough (today, for every ``anthropic/*`` id) or GMI/Vertex (planned once GMI is back). The +native Messages wire is the better transport, but on the OpenRouter path it re-writes the previous +turn's prompt cache on 14-20% of consecutive calls in concurrent tool loops (measured 2026-09-06; +NousResearch/api#227), so the session must ride chat/completions there. On GMI that is untested, +and until it is measured ``auto`` never promotes to native. + +The upstream IS visible in the first RESPONSE: OpenRouter stamps ``provider`` (chat wire) and +mints ``gen--`` ids; GMI/Vertex responses carry neither. So ``auto`` starts every +session on chat (safe on both upstreams), reads the first response, and switches the session to +native only when the upstream is GMI and native has been cleared for GMI. One decision per +session, at call 1, before there is a cache to lose; later calls never flip. + +``classify_upstream`` is pure and unit-tested; ``maybe_switch_wire_after_first_response`` is +the single hook, called from the usage recorder. +""" +from __future__ import annotations + +import logging +import re +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +# Flip to True only after the 20x6 concurrency probe (evals/postmortem/live_ab) is clean on a +# GMI-served anthropic/* id on the native wire. Until then ``auto`` is chat everywhere. +GMI_NATIVE_WIRE_CLEARED = False + +_OPENROUTER_ID = re.compile(r"^gen-\d{9,}-[A-Za-z0-9_-]{8,}$") + + +def classify_upstream(response: Any) -> Optional[str]: + """``"openrouter"`` / ``"gmi"`` / ``None`` (unknown) from a Portal response object. + + Works on both wires: the OpenAI SDK object exposes ``.provider`` (OpenRouter's upstream name, + e.g. ``"Anthropic"``, ``"Amazon Bedrock"``) and an OpenRouter-minted ``.id``; the Anthropic SDK + object has ``.id`` only. GMI/Vertex responses have Anthropic-native ``msg_…`` ids and no + ``provider``. Anything else is unknown, and unknown never triggers a switch. + """ + if response is None: + return None + if isinstance(getattr(response, "provider", None), str) and getattr(response, "provider"): + return "openrouter" + rid = getattr(response, "id", None) + if isinstance(rid, str): + if _OPENROUTER_ID.match(rid): + return "openrouter" + if rid.startswith("msg_"): + return "gmi" + return None + + +def wire_for_upstream(upstream: Optional[str]) -> str: + """The api_mode ``auto`` wants once the upstream is known. Chat unless GMI and cleared.""" + if upstream == "gmi" and GMI_NATIVE_WIRE_CLEARED: + return "anthropic_messages" + return "chat_completions" + + +def maybe_switch_wire_after_first_response(agent: Any, response: Any, api_call_count: int) -> bool: + """Decide the session's wire from its first response; the switch itself is applied at the + start of the next iteration (``apply_pending_wire_switch``), never while a response is being + consumed. Returns True when a switch was scheduled. + + Only for provider=nous, anthropic/* models, ``nous.anthropic_wire: auto``, and only on the + session's first API call. + """ + if api_call_count != 1 or getattr(agent, "_nous_wire_decided", False): + return False + if (getattr(agent, "provider", "") or "").lower() != "nous": + return False + model = str(getattr(agent, "model", "") or "") + if not model.lower().startswith("anthropic/"): + return False + try: + from hermes_cli.providers import _nous_anthropic_wire + if _nous_anthropic_wire() != "auto": + return False + except Exception: + return False + agent._nous_wire_decided = True # one decision per session, whatever it is + upstream = classify_upstream(response) + want = wire_for_upstream(upstream) + if want == getattr(agent, "api_mode", None): + logger.debug("nous wire auto: upstream=%s, staying on %s", upstream, want) + return False + agent._nous_wire_pending = (want, upstream) + return True + + +def apply_pending_wire_switch(agent: Any) -> bool: + """At iteration start, with no response in flight: perform the switch scheduled by + ``maybe_switch_wire_after_first_response``. Reuses ``switch_model`` (same model/provider, new + api_mode) so client rebuild, cache policy and ``_primary_runtime`` stay consistent. A failure + is logged and the session stays on its current wire.""" + pending = getattr(agent, "_nous_wire_pending", None) + if not pending: + return False + agent._nous_wire_pending = None + want, upstream = pending + try: + from agent.agent_runtime_helpers import switch_model + switch_model(agent, agent.model, "nous", api_key=getattr(agent, "api_key", "") or "", + base_url=getattr(agent, "base_url", "") or "", api_mode=want) + except Exception as exc: # never let wire selection break a turn + logger.warning("nous wire auto: switch to %s failed (%s); staying on %s", want, exc, agent.api_mode) + return False + logger.info("nous wire auto: upstream=%s -> %s for the rest of session %s", upstream, want, + getattr(agent, "session_id", "?")) + return True diff --git a/agent/turn_iteration_prep.py b/agent/turn_iteration_prep.py index 1b4917758c..96ac96f17c 100644 --- a/agent/turn_iteration_prep.py +++ b/agent/turn_iteration_prep.py @@ -39,6 +39,12 @@ def prepare_iteration(agent: Any,*, messages: Any, api_call_count: Any) -> Itera _INTERRUPT_SCAFFOLD_MARKER, _maybe_inject_run_budget_wrapup ) + # nous.anthropic_wire=auto: a wire switch decided from the previous response lands here, + # before this iteration's request is built and with nothing in flight. + if getattr(agent, "_nous_wire_pending", None): + from agent.nous_wire import apply_pending_wire_switch + apply_pending_wire_switch(agent) + # Fire step_callback for gateway hooks (agent:step event). if agent.step_callback is not None: try: diff --git a/agent/turn_usage.py b/agent/turn_usage.py index 131e2cd348..a535a6f937 100644 --- a/agent/turn_usage.py +++ b/agent/turn_usage.py @@ -181,6 +181,11 @@ def record_response_usage( prompt_tokens, completion_tokens, total_tokens, api_duration, _cache_pct, ) + # nous.anthropic_wire=auto: the session's wire is decided once, from this first response. + if agent.session_api_calls == 1 and (agent.provider or "") == "nous": + with suppress(Exception): + from agent.nous_wire import maybe_switch_wire_after_first_response + maybe_switch_wire_after_first_response(agent, response, agent.session_api_calls) # MoA: agent.model/provider are the virtual preset/"moa" with no pricing entry, silently # dropping aggregator spend. Price at the REAL model/provider from the aggregator slot. diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 12e08a5ecb..5f15909118 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -2286,10 +2286,12 @@ DEFAULT_CONFIG = { "keepalive_interval_seconds": 900, # anthropic_wire: which Portal route carries anthropic/* models. "chat" = # /v1/chat/completions (default for now); "native" = /v1/messages, the Anthropic - # Messages wire (signed thinking passthrough, native cache_control scopes). Native is the - # better wire but re-writes the previous turn's cache on 14-20% of consecutive calls in - # concurrent tool loops (measured 2026-09-06; NousResearch/api#227), so chat is the - # default until that is fixed. + # Messages wire (signed thinking passthrough, native cache_control scopes); "auto" = + # start on chat and, per session, switch to native from the first response when the + # Portal upstream serving the model is one where native is known clean. Native is the + # better wire but on the OpenRouter-served path it re-writes the previous turn's cache on + # 14-20% of consecutive calls in concurrent tool loops (measured 2026-09-06; + # NousResearch/api#227), so chat is the default until that is fixed. "anthropic_wire": "chat", }, # Google Vertex AI (Gemini). Auth is OAuth2 from a service-account JSON or ADC, NOT an API key; diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index dbec5d698a..69be630669 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -322,18 +322,21 @@ def nous_api_mode(model: str = "") -> str: signed native blocks, and cache_control scopes are translated by the portal's adapter. Empty/unknown model defaults to ``chat_completions`` (the historical Nous transport).""" if str(model or "").strip().lower().startswith("anthropic/"): + # ``auto`` starts on chat too: it is safe on every upstream, and ``agent/nous_wire.py`` + # promotes the session to native from the first response when the upstream allows it. return "anthropic_messages" if _nous_anthropic_wire() == "native" else "chat_completions" return "chat_completions" def _nous_anthropic_wire() -> str: - """``nous.anthropic_wire``: ``"chat"`` (default) or ``"native"``. Anything else reads as ``chat``.""" + """``nous.anthropic_wire``: ``"chat"`` (default), ``"native"``, or ``"auto"`` (chat, then per-session + promotion decided from the first response; see ``agent/nous_wire.py``). Anything else reads as ``chat``.""" try: from hermes_cli.config import load_config_readonly value = str(((load_config_readonly().get("nous") or {}).get("anthropic_wire")) or "chat").strip().lower() except Exception: return "chat" - return "native" if value == "native" else "chat" + return value if value in ("native", "auto") else "chat" def determine_api_mode(provider: str, base_url: str = "", model: str = "") -> str: diff --git a/tests/agent/test_nous_wire_auto.py b/tests/agent/test_nous_wire_auto.py new file mode 100644 index 0000000000..a5e17604c2 --- /dev/null +++ b/tests/agent/test_nous_wire_auto.py @@ -0,0 +1,139 @@ +"""``nous.anthropic_wire: auto`` decides a session's wire once, from its first response. + +Contracts: +- the upstream classifier reads what Portal actually returns on both wires (OpenRouter stamps + ``provider`` + ``gen-…`` ids; GMI/Vertex returns Anthropic-native ``msg_…`` ids, no provider); +- ``auto`` starts on chat and never promotes to native until GMI native is cleared; +- one decision per session, only on call 1, only for nous + anthropic/*, never for chat/native; +- through a real AIAgent and the real usage recorder, the hook fires exactly once and a switch + failure never breaks the turn. +""" +from __future__ import annotations + +from types import SimpleNamespace + +import pytest + +from agent import nous_wire +from hermes_cli import providers as _providers + + +def _resp(**kw): + return SimpleNamespace(**kw) + + +class TestClassifier: + def test_openrouter_by_provider_field_or_gen_id(self): + assert nous_wire.classify_upstream(_resp(provider="Anthropic", id="gen-1788708403-xXjDsFw")) == "openrouter" + assert nous_wire.classify_upstream(_resp(provider="Claude Platform on AWS", id="x")) == "openrouter" + # native wire: no provider field, but the id is still OpenRouter-minted + assert nous_wire.classify_upstream(_resp(id="gen-1788680469-hyks5Nj9n7rmdAbMyQ5x")) == "openrouter" + + def test_gmi_by_anthropic_native_id_without_provider(self): + assert nous_wire.classify_upstream(_resp(id="msg_01XrffvmxsRUWwCbVpBsADHo")) == "gmi" + + @pytest.mark.parametrize("r", [None, _resp(), _resp(id=None), _resp(id="chatcmpl-abc"), _resp(id="", provider="")]) + def test_unknown_never_classifies(self, r): + assert nous_wire.classify_upstream(r) is None + + +class TestWireChoice: + def test_chat_for_openrouter_and_unknown(self): + assert nous_wire.wire_for_upstream("openrouter") == "chat_completions" + assert nous_wire.wire_for_upstream(None) == "chat_completions" + + def test_gmi_is_chat_until_cleared_then_native(self, monkeypatch): + monkeypatch.setattr(nous_wire, "GMI_NATIVE_WIRE_CLEARED", False) + assert nous_wire.wire_for_upstream("gmi") == "chat_completions" + monkeypatch.setattr(nous_wire, "GMI_NATIVE_WIRE_CLEARED", True) + assert nous_wire.wire_for_upstream("gmi") == "anthropic_messages" + + +def _agent(**kw): + a = SimpleNamespace(provider="nous", model="anthropic/claude-fable-5.1", api_mode="chat_completions", + api_key="k", base_url="https://inference-api.nousresearch.com/v1", session_id="s") + for k, v in kw.items(): + setattr(a, k, v) + return a + + +class TestHook: + @pytest.fixture(autouse=True) + def _auto(self, monkeypatch): + monkeypatch.setattr(_providers, "_nous_anthropic_wire", lambda: "auto") + self.switches = [] + monkeypatch.setattr("agent.agent_runtime_helpers.switch_model", + lambda agent, m, p, api_key="", base_url="", api_mode="", **k: self.switches.append(api_mode) or setattr(agent, "api_mode", api_mode)) + + def test_gmi_cleared_schedules_once_and_applies_at_next_iteration(self, monkeypatch): + monkeypatch.setattr(nous_wire, "GMI_NATIVE_WIRE_CLEARED", True) + a = _agent() + assert nous_wire.maybe_switch_wire_after_first_response(a, _resp(id="msg_01abc"), 1) is True + # decided but NOT switched yet: the response is still being consumed on the old wire + assert a.api_mode == "chat_completions" and self.switches == [] + assert nous_wire.apply_pending_wire_switch(a) is True + assert a.api_mode == "anthropic_messages" and self.switches == ["anthropic_messages"] + assert nous_wire.apply_pending_wire_switch(a) is False # nothing pending twice + # a later call, even with a different-looking response, never flips again + assert nous_wire.maybe_switch_wire_after_first_response(a, _resp(provider="Anthropic", id="gen-1-x"), 2) is False + assert nous_wire.maybe_switch_wire_after_first_response(a, _resp(provider="Anthropic", id="gen-1-x"), 1) is False + assert self.switches == ["anthropic_messages"] + + def test_openrouter_stays_on_chat(self): + a = _agent() + assert nous_wire.maybe_switch_wire_after_first_response(a, _resp(provider="Anthropic", id="gen-1-x"), 1) is False + assert a.api_mode == "chat_completions" and self.switches == [] + + def test_gmi_uncleared_stays_on_chat(self, monkeypatch): + monkeypatch.setattr(nous_wire, "GMI_NATIVE_WIRE_CLEARED", False) + a = _agent() + assert nous_wire.maybe_switch_wire_after_first_response(a, _resp(id="msg_01abc"), 1) is False + assert a.api_mode == "chat_completions" and self.switches == [] + + @pytest.mark.parametrize("mode", ["chat", "native"]) + def test_explicit_modes_never_auto_switch(self, monkeypatch, mode): + monkeypatch.setattr(_providers, "_nous_anthropic_wire", lambda: mode) + monkeypatch.setattr(nous_wire, "GMI_NATIVE_WIRE_CLEARED", True) + a = _agent(api_mode="chat_completions" if mode == "chat" else "anthropic_messages") + assert nous_wire.maybe_switch_wire_after_first_response(a, _resp(id="msg_01abc"), 1) is False + assert self.switches == [] + + def test_other_providers_and_models_untouched(self, monkeypatch): + monkeypatch.setattr(nous_wire, "GMI_NATIVE_WIRE_CLEARED", True) + assert nous_wire.maybe_switch_wire_after_first_response(_agent(provider="openrouter"), _resp(id="msg_01abc"), 1) is False + assert nous_wire.maybe_switch_wire_after_first_response(_agent(model="openai/gpt-5.6-sol"), _resp(id="msg_01abc"), 1) is False + assert self.switches == [] + + def test_switch_failure_is_swallowed_and_decision_is_final(self, monkeypatch): + monkeypatch.setattr(nous_wire, "GMI_NATIVE_WIRE_CLEARED", True) + + def boom(*a, **k): + raise RuntimeError("no client") + monkeypatch.setattr("agent.agent_runtime_helpers.switch_model", boom) + a = _agent() + assert nous_wire.maybe_switch_wire_after_first_response(a, _resp(id="msg_01abc"), 1) is True + assert nous_wire.apply_pending_wire_switch(a) is False + assert a.api_mode == "chat_completions" and a._nous_wire_decided is True and a._nous_wire_pending is None + + +def test_real_agent_usage_recorder_calls_the_hook_once(tmp_path, monkeypatch): + """The wiring: record_response_usage on a real AIAgent invokes the hook on call 1 only.""" + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + (tmp_path / "config.yaml").write_text("nous:\n anthropic_wire: auto\n", encoding="utf-8") + from run_agent import AIAgent + from agent import turn_usage + calls = [] + monkeypatch.setattr(nous_wire, "maybe_switch_wire_after_first_response", + lambda agent, response, n: calls.append((n, nous_wire.classify_upstream(response))) or False) + a = AIAgent(api_key="jwt", base_url="https://inference-api.nousresearch.com/v1", provider="nous", + api_mode="chat_completions", model="anthropic/claude-fable-5.1", session_id="t", platform="cli", + quiet_mode=True, skip_context_files=True, skip_memory=True, save_trajectories=False, enabled_toolsets=["file"]) + try: + usage = SimpleNamespace(prompt_tokens=100, completion_tokens=5, total_tokens=105, prompt_tokens_details=None, completion_tokens_details=None) + for n in (1, 2, 3): + resp = SimpleNamespace(usage=usage, id="gen-1788708403-xXjDsFwabc", provider="Anthropic", model="anthropic/claude-fable-5.1") + turn_usage.record_response_usage(a, resp, messages=[{"role": "user", "content": "hi"}], api_call_count=n, + api_duration=0.1, compression_attempts=0, max_compression_attempts=3) + finally: + a.close() + assert calls == [(1, "openrouter")] diff --git a/website/docs/user-guide/configuring-models.md b/website/docs/user-guide/configuring-models.md index b9683f2826..2378ea231a 100644 --- a/website/docs/user-guide/configuring-models.md +++ b/website/docs/user-guide/configuring-models.md @@ -240,10 +240,12 @@ Nous Portal serves its `anthropic/*` models on two routes: OpenAI-compatible `/v ```yaml nous: - anthropic_wire: chat # default. "native" = the Anthropic Messages wire + anthropic_wire: chat # default. "native" = the Anthropic Messages wire; "auto" = decide per session ``` -`chat` is the default for now. The native wire is the better transport (signed thinking blocks pass through unchanged, native `cache_control` scopes), but in concurrent tool loops it currently re-writes the previous turn's prompt cache on 14–20% of consecutive calls, which is 15–20% of a fan-out's cache-write bill; the chat route measured 0 on the same test. Set `native` to opt back in (for example once the portal-side fix has shipped). Only `anthropic/*` models are affected; everything else on Nous already uses chat/completions. +`chat` is the default for now. The native wire is the better transport (signed thinking blocks pass through unchanged, native `cache_control` scopes), but on the Portal's OpenRouter-served path it currently re-writes the previous turn's prompt cache on 14–20% of consecutive calls in concurrent tool loops, which is 15–20% of a fan-out's cache-write bill; the chat route measured 0 on the same test. Set `native` to opt back in (for example once the portal-side fix has shipped). Only `anthropic/*` models are affected; everything else on Nous already uses chat/completions. + +`auto` is for when the Portal serves the same model from more than one upstream. A session starts on chat, Hermes reads which upstream answered the first call, and switches that session to native only when the upstream is one where native is known to be clean (the switch happens between calls, so no in-flight response and no warm cache is lost). Today no upstream is cleared, so `auto` behaves exactly like `chat`; it exists so the flip can be made from a measurement rather than a config change. ## When does it take effect? From 8cb2bcc8c147161d077e337f4d1afd89d82c16c2 Mon Sep 17 00:00:00 2001 From: Jerry Gooch Date: Sat, 5 Sep 2026 13:09:07 -0500 Subject: [PATCH 179/276] feat(desktop): expose structured session controls (cherry picked from commit 0dae0d9e7a36b57bc0489fbf0d6089de4f025d16) --- tests/tui_gateway/test_session_control.py | 499 ++++++++++++++++++++++ tui_gateway/methods_session_control.py | 388 +++++++++++++++++ tui_gateway/server.py | 6 +- 3 files changed, 891 insertions(+), 2 deletions(-) create mode 100644 tests/tui_gateway/test_session_control.py create mode 100644 tui_gateway/methods_session_control.py diff --git a/tests/tui_gateway/test_session_control.py b/tests/tui_gateway/test_session_control.py new file mode 100644 index 0000000000..549bc7fcf6 --- /dev/null +++ b/tests/tui_gateway/test_session_control.py @@ -0,0 +1,499 @@ +"""Behavioral contract tests for Desktop session-control JSON-RPC methods.""" + +from __future__ import annotations + +import importlib +import json +import threading +import time +import uuid +from types import SimpleNamespace +from unittest.mock import MagicMock, patch + +import pytest + + +@pytest.fixture() +def hermes_home(tmp_path, monkeypatch): + """Give the persisted control managers an isolated database for every test.""" + home = tmp_path / ".hermes" + home.mkdir() + monkeypatch.setattr("pathlib.Path.home", lambda: tmp_path) + monkeypatch.setenv("HERMES_HOME", str(home)) + from hermes_cli import goals + + goals._DB_CACHE.clear() + yield home + goals._DB_CACHE.clear() + + +@pytest.fixture() +def server(hermes_home, monkeypatch): + with patch.dict( + "sys.modules", + { + "hermes_cli.env_loader": MagicMock(), + "hermes_cli.banner": MagicMock(), + }, + ): + mod = importlib.import_module("tui_gateway.server") + monkeypatch.setattr(mod, "_hermes_home", hermes_home) + monkeypatch.setattr(mod, "_cfg_cache", None) + monkeypatch.setattr(mod, "_cfg_mtime", None) + monkeypatch.setattr(mod, "_cfg_path", None) + yield mod + mod._sessions.clear() + mod._pending.clear() + mod._answers.clear() + + +@pytest.fixture() +def session(server): + sid = f"sid-control-{uuid.uuid4().hex}" + key = f"control-{uuid.uuid4().hex}" + entry = { + "session_key": key, + "history": [], + "history_lock": threading.Lock(), + "history_version": 0, + "running": False, + "attached_images": [], + "cols": 120, + "agent": None, + "created_at": time.time(), + } + server._sessions[sid] = entry + yield sid, key, entry + from hermes_cli.goals import GoalManager + from hermes_cli.heartbeat import HeartbeatManager + from hermes_cli.loops import LoopManager + + GoalManager(key).clear() + LoopManager(key).clear() + HeartbeatManager(key).clear() + + +def _call(server, method, *, rid=91, **params): + return server._methods[method](rid, params) + + +def _control(server, sid): + return _call(server, "session.control.read", session_id=sid)["result"]["control"] + + +def _error(response): + assert "error" in response + return response["error"] + + +def _observe_dispatch(server, monkeypatch): + calls = [] + original = server._methods["command.dispatch"] + + def observe(rid, params): + calls.append((rid, dict(params))) + return original(rid, params) + + monkeypatch.setitem(server._methods, "command.dispatch", observe) + return calls + + +def _forbid_dispatch(server, monkeypatch): + def forbidden(_rid, _params): + raise AssertionError("manager-only action must not call command.dispatch") + + monkeypatch.setitem(server._methods, "command.dispatch", forbidden) + + +def _save_goal(key, **overrides): + from hermes_cli.goals import GoalState, save_goal + + fields = { + "goal": "Finish the desktop control card", + "status": "active", + "turns_used": 3, + "max_turns": 12, + "created_at": 100.0, + "last_turn_at": 200.0, + } + fields.update(overrides) + state = GoalState(**fields) + save_goal(key, state) + return state + + +def _save_loop(key, **overrides): + from hermes_cli.loops import LoopState, save_loop + + fields = { + "prompt": "Check the deployment", + "status": "active", + "mode": "interval", + "interval_seconds": 300, + "current_delay": 300, + "created_at": 100.0, + "next_due_at": 400.0, + } + fields.update(overrides) + state = LoopState(**fields) + save_loop(key, state) + return state + + +def _save_heartbeat(key, **overrides): + from hermes_cli.heartbeat import HeartbeatState, save_heartbeat + + fields = { + "prompt": "Check the deployment", + "interval_seconds": 600, + "status": "active", + "created_at": 100.0, + "last_fired_at": 150.0, + "fire_count": 2, + } + fields.update(overrides) + state = HeartbeatState(**fields) + save_heartbeat(key, state) + return state + + +class TestStructuredRead: + def test_methods_are_registered_and_empty_snapshot_is_stable(self, server, session): + sid, _, _ = session + assert {"session.control.read", "session.control"} <= set(server._methods) + first = _control(server, sid) + second = _control(server, sid) + assert first == second + assert first == { + "goal": None, + "loop": None, + "heartbeat": None, + "revision": "", + "updated_at": 0, + } + + def test_goal_contract_subgoals_and_gates_are_structured_and_sanitized(self, server, session): + from hermes_cli.goals import GoalContract, GoalGate + + sid, key, _ = session + contract = GoalContract(outcome="Card is correct", verification="Run focused tests") + _save_goal( + key, + contract=contract, + subgoals=["Keep command routing narrow", "Document event hydration seam"], + gates=[GoalGate( + command="scripts/run_tests.sh tests/tui_gateway/test_session_control.py", + timeout_seconds=90, + max_retries=2, + attempts=1, + last_exit_code=1, + last_output_tail="private output must stay private", + last_failed_fingerprint="secret-fingerprint", + )], + ) + + goal = _control(server, sid)["goal"] + assert goal["title"] == "Finish the desktop control card" + assert goal["contract"] == contract.to_dict() + assert goal["subgoals"] == ["Keep command routing narrow", "Document event hydration seam"] + assert goal["gates"] == [{ + "command": "scripts/run_tests.sh tests/tui_gateway/test_session_control.py", + "timeout_seconds": 90, + "max_retries": 2, + "attempts": 1, + "last_exit_code": 1, + }] + serialized = json.dumps(goal) + for forbidden in ( + "last_output_tail", "last_failed_fingerprint", "private output", "secret-fingerprint", + "route", "session_id", "credential", "api_key", + ): + assert forbidden not in serialized + + def test_wait_barrier_is_absolute_and_unchanged_reads_do_not_count_down(self, server, session, monkeypatch): + sid, key, _ = session + _save_goal(key, waiting_until=2_000.0, waiting_reason="rate limit") + clock = SimpleNamespace(now=1_000.0) + monkeypatch.setattr(server, "time", SimpleNamespace(time=lambda: clock.now)) + + first = _control(server, sid) + clock.now = 1_001.0 + second = _control(server, sid) + + assert first == second + assert first["goal"]["wait_barrier"] == { + "type": "until", "until_at": 2_000.0, "reason": "rate limit", + } + assert "remaining_seconds" not in first["goal"]["wait_barrier"] + + @pytest.mark.parametrize( + ("kind", "setup"), + [ + ("goal", lambda key: _save_goal(key)), + ("loop", lambda key: _save_loop(key)), + ("heartbeat", lambda key: _save_heartbeat(key)), + ], + ) + def test_revision_is_stable_then_changes_for_each_visible_control(self, server, session, kind, setup): + sid, key, _ = session + setup(key) + before = _control(server, sid) + assert isinstance(before["revision"], str) + assert len(before["revision"]) == 64 + assert before == _control(server, sid) + + if kind == "goal": + _save_goal(key, goal="A visible replacement goal") + elif kind == "loop": + _save_loop(key, prompt="A visibly changed loop prompt") + else: + _save_heartbeat(key, prompt="A visibly changed heartbeat prompt") + + assert _control(server, sid)["revision"] != before["revision"] + + def test_session_and_pid_wait_barriers_keep_absolute_targets(self, server, session): + sid, key, _ = session + _save_goal(key, waiting_on_session="bg-session", waiting_reason="CI") + assert _control(server, sid)["goal"]["wait_barrier"] == { + "type": "session", "target": "bg-session", "reason": "CI", + } + _save_goal(key, waiting_on_pid=4242, waiting_reason="build") + assert _control(server, sid)["goal"]["wait_barrier"] == { + "type": "pid", "target": 4242, "reason": "build", + } + + +@pytest.mark.parametrize( + ("action", "name", "arg", "status"), + [ + ("goal.pause", "goal", "pause", "active"), + ("goal.resume", "goal", "resume", "paused"), + ("goal.clear", "goal", "clear", "active"), + ("loop.pause", "loop", "pause", "active"), + ("loop.resume", "loop", "resume", "paused"), + ("loop.stop", "loop", "stop", "active"), + ], +) +def test_dispatched_actions_use_exact_mapping_and_caller_rpc_id( + server, session, monkeypatch, action, name, arg, status, +): + sid, key, _ = session + if name == "goal": + _save_goal(key, status=status) + else: + _save_loop(key, status=status) + calls = _observe_dispatch(server, monkeypatch) + + response = _call(server, "session.control", rid=743, session_id=sid, action=action) + + assert "result" in response + assert calls == [(743, {"session_id": sid, "name": name, "arg": arg})] + + +class TestDispatcherBackedMutations: + def test_goal_pause_resume_clear_mutate_real_persisted_state(self, server, session): + sid, key, _ = session + _save_goal(key, status="active", turns_used=8) + + paused = _call(server, "session.control", session_id=sid, action="goal.pause") + assert paused["result"]["control"]["goal"]["status"] == "paused" + + resumed = _call(server, "session.control", session_id=sid, action="goal.resume") + assert resumed["result"]["dispatch"]["type"] == "send" + assert resumed["result"]["dispatch"]["message"] + assert resumed["result"]["dispatch"]["notice"] + assert resumed["result"]["dispatch"]["display"] == "/goal resume" + assert resumed["result"]["control"]["goal"]["status"] == "active" + + cleared = _call(server, "session.control", session_id=sid, action="goal.clear") + assert cleared["result"]["control"]["goal"] is None + + def test_loop_pause_resume_stop_mutate_real_persisted_state(self, server, session): + sid, key, _ = session + _save_loop(key) + + paused = _call(server, "session.control", session_id=sid, action="loop.pause") + assert paused["result"]["control"]["loop"]["status"] == "paused" + + resumed = _call(server, "session.control", session_id=sid, action="loop.resume") + assert resumed["result"]["control"]["loop"]["status"] == "active" + + stopped = _call(server, "session.control", session_id=sid, action="loop.stop") + assert stopped["result"]["control"]["loop"] is None + + +class TestManagerOnlyMutations: + def test_subgoal_add_remove_clear_are_real_one_based_mutations_without_dispatch(self, server, session, monkeypatch): + from hermes_cli.goals import load_goal + + sid, key, _ = session + _save_goal(key, subgoals=["First criterion"]) + _forbid_dispatch(server, monkeypatch) + + added = _call(server, "session.control", session_id=sid, action="subgoal.add", args={"text": "Second criterion"}) + assert added["result"]["dispatch"]["output"] == "✓ Added subgoal 2: Second criterion" + assert load_goal(key).subgoals == ["First criterion", "Second criterion"] + + removed = _call(server, "session.control", session_id=sid, action="subgoal.remove", args={"index": 1}) + assert removed["result"]["dispatch"]["output"] == "✓ Removed subgoal 1: First criterion" + assert load_goal(key).subgoals == ["Second criterion"] + + cleared = _call(server, "session.control", session_id=sid, action="subgoal.clear") + assert cleared["result"]["dispatch"]["output"] == "✓ Cleared 1 subgoal." + assert load_goal(key).subgoals == [] + + def test_subgoal_requires_a_goal_and_valid_arguments_without_dispatch(self, server, session, monkeypatch): + sid, key, _ = session + _forbid_dispatch(server, monkeypatch) + assert _error(_call(server, "session.control", session_id=sid, action="subgoal.add", args={"text": "criterion"}))["code"] == 4004 + + _save_goal(key, subgoals=["Only criterion"]) + for args in ( + {"text": " "}, + {"index": "1"}, + {"index": 1.5}, + {"index": True}, + {"index": 0}, + {"index": 2}, + ): + action = "subgoal.add" if "text" in args else "subgoal.remove" + assert _error(_call(server, "session.control", session_id=sid, action=action, args=args))["code"] == 4004 + + def test_goal_unwait_clears_the_real_barrier_without_dispatch(self, server, session, monkeypatch): + from hermes_cli.goals import GoalManager + + sid, key, _ = session + _save_goal(key) + GoalManager(key).wait_for_seconds(60, reason="backoff") + _forbid_dispatch(server, monkeypatch) + + response = _call(server, "session.control", session_id=sid, action="goal.unwait") + assert response["result"]["dispatch"]["output"] == "▶ Wait barrier cleared — goal loop resumes." + assert GoalManager(key).state.waiting_until == 0.0 + + def test_heartbeat_pause_resume_clear_and_no_heartbeat_messages_do_not_dispatch(self, server, session, monkeypatch): + sid, key, _ = session + _forbid_dispatch(server, monkeypatch) + assert _call(server, "session.control", session_id=sid, action="heartbeat.pause")["result"]["dispatch"]["output"] == "No heartbeat set." + assert _call(server, "session.control", session_id=sid, action="heartbeat.resume")["result"]["dispatch"]["output"] == "No heartbeat to resume." + assert _call(server, "session.control", session_id=sid, action="heartbeat.clear")["result"]["dispatch"]["output"] == "No heartbeat set." + + _save_heartbeat(key, status="active", last_fired_at=1.0) + paused = _call(server, "session.control", session_id=sid, action="heartbeat.pause") + assert paused["result"]["dispatch"]["output"] == "⏸ Heartbeat paused: Check the deployment" + assert paused["result"]["control"]["heartbeat"]["status"] == "paused" + + resumed = _call(server, "session.control", session_id=sid, action="heartbeat.resume") + assert resumed["result"]["dispatch"]["output"] == "▶ Heartbeat resumed (every 10m): Check the deployment" + assert resumed["result"]["control"]["heartbeat"]["last_fired_at"] > 1.0 + + cleared = _call(server, "session.control", session_id=sid, action="heartbeat.clear") + assert cleared["result"]["dispatch"]["output"] == "✓ Heartbeat cleared." + assert cleared["result"]["control"]["heartbeat"] is None + + +class TestErrorsAndEvents: + @pytest.mark.parametrize( + ("action", "args"), + [ + ("", {}), + ("not.allowed", {}), + ("goal.gate.add", {"command": "echo should-not-run"}), + ("subgoal.add", []), + ], + ) + def test_invalid_actions_and_malformed_args_return_4004_without_dispatch(self, server, session, monkeypatch, action, args): + sid, _, _ = session + _forbid_dispatch(server, monkeypatch) + assert _error(_call(server, "session.control", session_id=sid, action=action, args=args))["code"] == 4004 + + def test_unknown_session_returns_4001(self, server): + assert _error(_call(server, "session.control", session_id="gone", action="goal.pause"))["code"] == 4001 + assert _error(_call(server, "session.control.read", session_id="gone"))["code"] == 4001 + + def test_dispatch_error_emits_no_update(self, server, session, monkeypatch): + sid, key, _ = session + _save_goal(key) + emitted = [] + monkeypatch.setattr(server, "_emit", lambda *event: emitted.append(event)) + monkeypatch.setitem(server._methods, "command.dispatch", lambda rid, params: server._err(rid, 4018, "dispatch failed")) + + response = _call(server, "session.control", session_id=sid, action="goal.pause") + assert _error(response)["code"] == 4018 + assert emitted == [] + + def test_adapter_error_becomes_4004_and_emits_no_update(self, server, session, monkeypatch): + from hermes_cli.goals import GoalManager + + sid, key, _ = session + _save_goal(key) + emitted = [] + monkeypatch.setattr(server, "_emit", lambda *event: emitted.append(event)) + monkeypatch.setattr(GoalManager, "add_subgoal", lambda self, text: (_ for _ in ()).throw(RuntimeError("blocked"))) + + response = _call(server, "session.control", session_id=sid, action="subgoal.add", args={"text": "criterion"}) + assert _error(response)["code"] == 4004 + assert emitted == [] + + def test_success_response_snapshot_is_exactly_the_one_update_event(self, server, session, monkeypatch): + sid, key, _ = session + _save_goal(key) + emitted = [] + monkeypatch.setattr(server, "_emit", lambda event, event_sid, payload=None: emitted.append((event, event_sid, payload))) + + response = _call(server, "session.control", session_id=sid, action="goal.pause") + assert emitted == [("session.control.update", sid, {"control": response["result"]["control"]})] + + def test_dispatch_envelope_preserves_all_user_visible_fields(self, server, session, monkeypatch): + sid, key, _ = session + _save_goal(key, status="paused") + expected = { + "type": "send", + "output": "already-rendered output", + "notice": "Goal resumed", + "message": "Continue toward the goal.", + "display": "/goal resume", + } + monkeypatch.setitem( + server._methods, + "command.dispatch", + lambda rid, params: {"jsonrpc": "2.0", "id": rid, "result": expected}, + ) + + response = _call(server, "session.control", session_id=sid, action="goal.resume") + assert response["result"]["dispatch"] == expected + + def test_emit_failure_must_not_turn_mutation_into_rpc_error(self, server, session, monkeypatch): + """Event delivery is best-effort: a failing _emit must not swallow an + already-applied mutation. The persisted goal must be paused, the RPC + response must carry the correct snapshot and dispatch envelope, and the + emission error must be debug-logged rather than propagated.""" + from hermes_cli.goals import GoalManager + + sid, key, _ = session + _save_goal(key, status="active") + monkeypatch.setattr( + server, "_emit", lambda *a, **kw: (_ for _ in ()).throw(RuntimeError("emit broken")), + ) + + response = _call(server, "session.control", session_id=sid, action="goal.pause") + + assert "result" in response, f"_emit failure leaked as RPC error: {response}" + assert response["result"]["control"]["goal"]["status"] == "paused" + assert response["result"]["dispatch"]["type"] == "exec" + assert response["result"]["dispatch"]["output"] + assert GoalManager(key).state.status == "paused" + + def test_snapshot_failure_after_mutation_does_not_emit_an_all_null_fallback(self, server, session, monkeypatch): + from hermes_cli.goals import GoalManager + + sid, key, _ = session + _save_goal(key) + emitted = [] + monkeypatch.setattr(server, "_emit", lambda *event: emitted.append(event)) + monkeypatch.setattr(server, "_snapshot_control", lambda _key: (_ for _ in ()).throw(RuntimeError("storage failed"))) + + response = _call(server, "session.control", session_id=sid, action="goal.pause") + assert _error(response)["code"] == 5031 + assert GoalManager(key).state.status == "paused" + assert emitted == [] diff --git a/tui_gateway/methods_session_control.py b/tui_gateway/methods_session_control.py new file mode 100644 index 0000000000..b1a7b7e12c --- /dev/null +++ b/tui_gateway/methods_session_control.py @@ -0,0 +1,388 @@ +"""Structured Desktop controls for persisted goal, loop, and heartbeat state. + +``session.control.read`` is a stable, allowlisted view of one live session. +``session.control`` accepts a closed set of intent-level actions: goal and +loop actions use their existing TUI command handlers, while controls without a +TUI command use the public manager API. A successful action emits one matching +``session.control.update`` event. +""" + +from __future__ import annotations + +import hashlib +import json +import logging +import time + +from .method_ctx import HandlerRegistry, bind_module + +_registry = HandlerRegistry() +method = _registry.method +_profile_scoped = _registry.profile_scoped + +logger = logging.getLogger(__name__) + + +_ACTION_COMMAND_MAP: dict[str, tuple[str, str]] = { + "goal.pause": ("goal", "pause"), + "goal.resume": ("goal", "resume"), + "goal.clear": ("goal", "clear"), + "loop.pause": ("loop", "pause"), + "loop.resume": ("loop", "resume"), + "loop.stop": ("loop", "stop"), +} + +_MANAGER_ACTIONS = frozenset({ + "goal.unwait", + "subgoal.add", + "subgoal.remove", + "subgoal.clear", + "heartbeat.pause", + "heartbeat.resume", + "heartbeat.clear", +}) + +_VALID_ACTIONS = frozenset(_ACTION_COMMAND_MAP) | _MANAGER_ACTIONS + + +def _safe_goal_snapshot(state) -> dict | None: + """Return only stable, frontend-safe GoalState fields.""" + if state is None or state.status == "cleared": + return None + snapshot = { + "title": state.goal, + "status": state.status, + "turns_used": state.turns_used, + "max_turns": state.max_turns, + "contract": state.contract.to_dict(), + "subgoals": list(state.subgoals), + "gates": [ + { + "command": gate.command, + "timeout_seconds": gate.timeout_seconds, + "max_retries": gate.max_retries, + "attempts": gate.attempts, + "last_exit_code": gate.last_exit_code, + } + for gate in state.gates + ], + } + if state.created_at: + snapshot["created_at"] = state.created_at + if state.last_turn_at: + snapshot["updated_at"] = state.last_turn_at + if state.paused_reason: + snapshot["paused_reason"] = state.paused_reason + if state.last_verdict: + snapshot["last_verdict"] = state.last_verdict + if state.last_reason: + snapshot["last_reason"] = state.last_reason + if barrier := _extract_wait_barrier(state): + snapshot["wait_barrier"] = barrier + return snapshot + + +def _extract_wait_barrier(state) -> dict | None: + """Expose absolute barriers; countdown presentation belongs to the client.""" + reason = state.waiting_reason or "" + if state.waiting_until and time.time() < state.waiting_until: + return {"type": "until", "until_at": state.waiting_until, "reason": reason} + if state.waiting_on_session is not None: + return {"type": "session", "target": state.waiting_on_session, "reason": reason} + if state.waiting_on_pid is not None: + return {"type": "pid", "target": state.waiting_on_pid, "reason": reason} + return None + + +def _safe_loop_snapshot(state, *, deferred_by_goal: bool) -> dict | None: + """Return allowlisted persisted LoopState fields, never its route.""" + if state is None or state.status == "cleared": + return None + snapshot = { + "prompt": state.prompt, + "status": state.status, + "mode": state.mode, + "interval_seconds": state.interval_seconds, + "current_delay": state.current_delay, + "times": state.times, + "until": state.until, + "max_ticks": state.max_ticks, + "ticks_fired": state.ticks_fired, + "created_at": state.created_at, + "last_fired_at": state.last_fired_at, + "next_due_at": state.next_due_at, + "awaiting_response": state.awaiting_response, + "deferred_by_goal": deferred_by_goal, + } + if state.paused_reason: + snapshot["paused_reason"] = state.paused_reason + if state.last_stop_reason: + snapshot["last_stop_reason"] = state.last_stop_reason + return snapshot + + +def _safe_heartbeat_snapshot(state) -> dict | None: + """Return only the persisted HeartbeatState fields the Desktop renders.""" + if state is None or state.status == "cleared": + return None + return { + "prompt": state.prompt, + "status": state.status, + "interval_seconds": state.interval_seconds, + "created_at": state.created_at, + "last_fired_at": state.last_fired_at, + "fire_count": state.fire_count, + } + + +def _snapshot_control(session_key: str) -> dict: + """Serialize persisted session-control state once, without wall-clock churn.""" + goal_state = _load_goal_state(session_key) + loop_state = _load_loop_state(session_key) + heartbeat_state = _load_heartbeat_state(session_key) + deferred_by_goal = bool( + loop_state is not None + and loop_state.status == "active" + and goal_state is not None + and goal_state.status == "active" + and _goal_blocks_loop_tick(session_key) + ) + goal = _safe_goal_snapshot(goal_state) + loop = _safe_loop_snapshot(loop_state, deferred_by_goal=deferred_by_goal) + heartbeat = _safe_heartbeat_snapshot(heartbeat_state) + return { + "goal": goal, + "loop": loop, + "heartbeat": heartbeat, + "revision": _snapshot_revision(goal, loop, heartbeat), + "updated_at": _snapshot_updated_at(goal_state, loop_state, heartbeat_state), + } + + +def _snapshot_revision(goal, loop, heartbeat) -> str: + """Hash canonical visible state so equal reads always have equal revisions.""" + if goal is None and loop is None and heartbeat is None: + return "" + canonical = json.dumps( + {"goal": goal, "loop": loop, "heartbeat": heartbeat}, + ensure_ascii=False, + separators=(",", ":"), + sort_keys=True, + ) + return hashlib.sha256(canonical.encode("utf-8")).hexdigest() + + +def _snapshot_updated_at(goal_state, loop_state, heartbeat_state): + """Use only persisted timestamps; reads never manufacture a new timestamp.""" + candidates = [] + for state, fields in ( + (goal_state, ("created_at", "last_turn_at")), + (loop_state, ("created_at", "last_fired_at")), + (heartbeat_state, ("created_at", "last_fired_at")), + ): + if state is not None: + candidates.extend(value for field in fields if (value := getattr(state, field, 0))) + return max(candidates) if candidates else 0 + + +def _load_goal_state(session_key): + from hermes_cli.goals import load_goal + + return load_goal(session_key) + + +def _load_loop_state(session_key): + from hermes_cli.loops import load_loop + + return load_loop(session_key) + + +def _load_heartbeat_state(session_key): + from hermes_cli.heartbeat import load_heartbeat + + return load_heartbeat(session_key) + + +def _goal_blocks_loop_tick(session_key: str) -> bool: + from hermes_cli.loops import goal_blocks_loop_tick + + return goal_blocks_loop_tick(session_key) + + +@method("session.control.read") +@_profile_scoped +def _(rid, params: dict) -> dict: + """Return the current stable control snapshot for a live session.""" + session, err = _sess_nowait(params, rid) + if err: + return err + session_key = str(session.get("session_key") or "") + if not session_key: + return _err(rid, 4001, "session has no stored key") + try: + return _ok(rid, {"control": _snapshot_control(session_key)}) + except Exception as exc: + logger.debug("session.control.read failed: %s", exc, exc_info=True) + return _err(rid, 5031, f"session.control.read failed: {exc}") + + +@method("session.control") +@_profile_scoped +def _(rid, params: dict) -> dict: + """Run one allowlisted control action and emit its exact resulting snapshot.""" + raw_action = params.get("action") + if not isinstance(raw_action, str) or not (action := raw_action.strip()): + return _err(rid, 4004, "action is required") + if action.startswith("goal.gate"): + return _err(rid, 4004, "gate actions are not allowed through session.control") + if action not in _VALID_ACTIONS: + return _err(rid, 4004, f"unknown action: {action}") + + if "args" in params: + args = params["args"] + if not isinstance(args, dict): + return _err(rid, 4004, "args must be an object") + else: + args = {} + validated, validation_error = _validate_action_args(rid, action, args) + if validation_error: + return validation_error + + session, err = _sess_nowait(params, rid) + if err: + return err + session_key = str(session.get("session_key") or "") + if not session_key: + return _err(rid, 4001, "session has no stored key") + + try: + if action in _ACTION_COMMAND_MAP: + name, arg = _ACTION_COMMAND_MAP[action] + action_result = _dispatch_command(rid, session_id=params.get("session_id") or "", name=name, arg=arg) + else: + action_result = _execute_manager_action(session_key, action, validated) + except (RuntimeError, ValueError, IndexError) as exc: + return _err(rid, 4004, _manager_error_message(action, exc)) + + if "error" in action_result: + return action_result + + try: + control = _snapshot_control(session_key) + except Exception as exc: + logger.debug("session.control snapshot after %s failed: %s", action, exc, exc_info=True) + return _err(rid, 5031, f"session.control snapshot failed: {exc}") + + try: + _emit("session.control.update", params.get("session_id") or "", {"control": control}) + except Exception as exc: + logger.debug("session.control.update emit failed (best-effort): %s", exc, exc_info=True) + return _ok(rid, {"control": control, "dispatch": _dispatch_envelope(action_result)}) + + +def _validate_action_args(rid, action: str, args: dict): + """Validate the only actions with input before any manager is constructed.""" + if action == "subgoal.add": + text = args.get("text") + if not isinstance(text, str) or not (text := text.strip()): + return None, _err(rid, 4004, "subgoal text is required") + return {"text": text}, None + if action == "subgoal.remove": + index = args.get("index") + if type(index) is not int: + return None, _err(rid, 4004, "subgoal index must be an integer") + if index < 1: + return None, _err(rid, 4004, "subgoal index must be >= 1") + return {"index": index}, None + return {}, None + + +def _dispatch_command(rid, *, session_id: str, name: str, arg: str) -> dict: + """Delegate a fixed intent to the existing TUI command dispatcher.""" + handler = _methods.get("command.dispatch") + if handler is None: + return _err(rid, 5031, "command.dispatch unavailable") + try: + return handler(rid, {"session_id": session_id, "name": name, "arg": arg}) + except Exception as exc: + logger.debug("command.dispatch %s %s failed: %s", name, arg, exc, exc_info=True) + return _err(rid, 5031, f"dispatch failed: {exc}") + + +def _execute_manager_action(session_key: str, action: str, args: dict) -> dict: + """Use manager APIs for controls that have no TUI command handler.""" + if action == "goal.unwait": + return _execute_goal_unwait(session_key) + if action.startswith("subgoal."): + return _execute_subgoal_action(session_key, action, args) + return _execute_heartbeat_action(session_key, action) + + +def _execute_goal_unwait(session_key: str) -> dict: + from hermes_cli.goals import GoalManager + + manager = GoalManager(session_id=session_key) + output = "▶ Wait barrier cleared — goal loop resumes." if manager.stop_waiting() else "No wait barrier set." + return {"result": {"type": "exec", "output": output}} + + +def _execute_subgoal_action(session_key: str, action: str, args: dict) -> dict: + from hermes_cli.goals import GoalManager + + manager = GoalManager(session_id=session_key) + if action == "subgoal.add": + text = manager.add_subgoal(args["text"]) + return {"result": {"type": "exec", "output": f"✓ Added subgoal {len(manager.state.subgoals)}: {text}"}} + if action == "subgoal.remove": + index = args["index"] + text = manager.remove_subgoal(index) + return {"result": {"type": "exec", "output": f"✓ Removed subgoal {index}: {text}"}} + count = manager.clear_subgoals() + output = f"✓ Cleared {count} subgoal{'s' if count != 1 else ''}." if count else "No subgoals to clear." + return {"result": {"type": "exec", "output": output}} + + +def _execute_heartbeat_action(session_key: str, action: str) -> dict: + from hermes_cli.heartbeat import HeartbeatManager, format_interval + + manager = HeartbeatManager(session_id=session_key) + if action == "heartbeat.pause": + state = manager.pause() + output = f"⏸ Heartbeat paused: {state.prompt}" if state else "No heartbeat set." + elif action == "heartbeat.resume": + state = manager.resume() + output = ( + f"▶ Heartbeat resumed (every {format_interval(state.interval_seconds)}): {state.prompt}" + if state else "No heartbeat to resume." + ) + elif action == "heartbeat.clear": + output = "✓ Heartbeat cleared." if manager.clear() else "No heartbeat set." + else: + return _err(None, 4004, f"unknown heartbeat action: {action}") + return {"result": {"type": "exec", "output": output}} + + +def _manager_error_message(action: str, exc: Exception) -> str: + prefixes = { + "subgoal.add": "/subgoal", + "subgoal.remove": "/subgoal remove", + "subgoal.clear": "/subgoal clear", + } + return f"{prefixes.get(action, action)}: {exc}" + + +def _dispatch_envelope(response: dict) -> dict: + """Keep the command result's user-visible envelope without adding model-facing data.""" + result = response.get("result") or {} + return { + "type": result.get("type"), + "output": result.get("output"), + "notice": result.get("notice"), + "message": result.get("message"), + "display": result.get("display"), + } + + +def register(server) -> None: + """Rebind this module's handlers onto the server namespace.""" + bind_module(globals(), server, skip=("_",)) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 33ebf7ba25..316ad72ba1 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -3207,7 +3207,8 @@ from . import ( # noqa: E402 methods_config_set as _methods_config_set, methods_images as _methods_images, methods_profiles as _methods_profiles, methods_prompt as _methods_prompt, methods_session as _methods_session, methods_tools as _methods_tools, prompt_turn as _prompt_turn, billing_view as _billing_view, - methods_projects as _methods_projects, methods_session_foreign as _methods_session_foreign) + methods_projects as _methods_projects, methods_session_foreign as _methods_session_foreign, + methods_session_control as _methods_session_control) for _m in ( _session_reaper, _session_lifecycle, _session_workdir, _compute_host_bridge, _model_switch, @@ -3216,6 +3217,7 @@ for _m in ( _methods_complete_helpers, _methods_slash, _methods_voice, _methods_browser, _methods_browser_control, _methods_session, _methods_prompt, _methods_config, _methods_config_set, _methods_complete, _methods_tools, _methods_profiles, _methods_images, - _methods_bot_relay, _prompt_turn, _billing_view, _methods_projects, _methods_session_foreign): + _methods_bot_relay, _prompt_turn, _billing_view, _methods_projects, _methods_session_foreign, + _methods_session_control): _m.register(sys.modules[__name__]) del _m From bfddf556bff49b7c419f133c798892bec36012a9 Mon Sep 17 00:00:00 2001 From: Jerry Gooch Date: Sat, 5 Sep 2026 13:59:41 -0500 Subject: [PATCH 180/276] feat(desktop): hydrate structured session controls (cherry picked from commit fec4bab6a191c474456956a860d86957d435552b) --- .../use-message-stream/gateway-event/index.ts | 2 + .../gateway-event/message-stream.test.ts | 55 ++ .../gateway-event/message-stream.ts | 5 + .../gateway-event/session-control.test.ts | 72 ++ .../gateway-event/session-control.ts | 20 + .../desktop/src/store/session-control.test.ts | 457 ++++++++++ apps/desktop/src/store/session-control.ts | 838 ++++++++++++++++++ 7 files changed, 1449 insertions(+) create mode 100644 apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.test.ts create mode 100644 apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/session-control.test.ts create mode 100644 apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/session-control.ts create mode 100644 apps/desktop/src/store/session-control.test.ts create mode 100644 apps/desktop/src/store/session-control.ts diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/index.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/index.ts index 8c86f65d81..572e5c138e 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/index.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/index.ts @@ -20,6 +20,7 @@ import { handleDesktopBridgeEvent } from './desktop-bridge' import { handleInputRequestEvent } from './input-requests' import { handleLifecycleEvent } from './lifecycle' import { handleMessageStreamEvent } from './message-stream' +import { handleControlEvent } from './session-control' import { handleSessionInfoEvent } from './session-info' import { handleStatusEvent } from './status' import { handleToolEvent } from './tools' @@ -84,6 +85,7 @@ const PROVIDER_WAIT_SUPERSEDING_EVENT_TYPES = new Set([ const HANDLERS: GatewayEventHandler[] = [ handleLifecycleEvent, handleSessionInfoEvent, + handleControlEvent, handleMessageStreamEvent, handleToolEvent, handleInputRequestEvent, diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.test.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.test.ts new file mode 100644 index 0000000000..db81acdab1 --- /dev/null +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.test.ts @@ -0,0 +1,55 @@ +import { describe, expect, it, vi } from 'vitest' + +const { refreshSupportedSessionControlAfterTurn } = vi.hoisted(() => ({ + refreshSupportedSessionControlAfterTurn: vi.fn(async () => undefined) +})) + +vi.mock('@/store/session-control', () => ({ refreshSupportedSessionControlAfterTurn })) + +import { handleMessageStreamEvent } from './message-stream' +import type { GatewayEventContext } from './types' + +function context(type: string): GatewayEventContext { + return { + deps: { + activeGatewayProfile: 'default', + activeSessionIdRef: { current: 's1' }, + appendAssistantDelta: vi.fn(), + appendReasoningDelta: vi.fn(), + compactedTurnRef: { current: new Set() }, + completeAssistantMessage: vi.fn(), + failAssistantMessage: vi.fn(), + finalizeInterimAssistantMessage: vi.fn(), + flushQueuedDeltas: vi.fn(), + hydrateFromStoredSession: vi.fn(async () => undefined), + lastCwdInfoSessionRef: { current: null }, + nativeSubagentSessionsRef: { current: new Set() }, + queryClient: {} as GatewayEventContext['deps']['queryClient'], + refreshHermesConfig: vi.fn(async () => undefined), + scheduleSessionsRefresh: vi.fn(), + sessionInterrupted: vi.fn(() => false), + sessionStateByRuntimeIdRef: { current: new Map() }, + updateSessionState: vi.fn(), + upsertToolCall: vi.fn() + }, + event: { type }, + explicitSid: 's1', + fromActiveSource: () => true, + isActiveEvent: false, + occurredAt: 1_700_000_100, + payload: { text: 'completed' }, + scheduleConfigRefresh: vi.fn(), + sessionId: 's1' + } +} + +describe('handleMessageStreamEvent session-control integration', () => { + it('refreshes only after message.complete, through the store seam', () => { + expect(handleMessageStreamEvent(context('message.delta'))).toBe(true) + expect(refreshSupportedSessionControlAfterTurn).not.toHaveBeenCalled() + + expect(handleMessageStreamEvent(context('message.complete'))).toBe(true) + expect(refreshSupportedSessionControlAfterTurn).toHaveBeenCalledTimes(1) + expect(refreshSupportedSessionControlAfterTurn).toHaveBeenCalledWith('s1') + }) +}) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts index 43be63e190..97fb4a65a5 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts @@ -14,6 +14,7 @@ import { flashPetActivity, markPetUnread, setPetActivity } from '@/store/pet' import { clearAllPrompts } from '@/store/prompts' import { providerWaitText, setSessionProviderWait } from '@/store/provider-wait' import { setCurrentUsage, setTurnStartedAt } from '@/store/session' +import { refreshSupportedSessionControlAfterTurn } from '@/store/session-control' import { pruneFinishedSessionSubagents } from '@/store/subagents' import { clearActiveSessionTodos } from '@/store/todos' @@ -388,6 +389,10 @@ export function handleMessageStreamEvent(ctx: GatewayEventContext): boolean { } } + // Refresh only the structured-control sessions already proven capable. + // Initial hydration owns the unknown capability probe. + void refreshSupportedSessionControlAfterTurn(sessionId) + return true } diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/session-control.test.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/session-control.test.ts new file mode 100644 index 0000000000..6211e57b71 --- /dev/null +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/session-control.test.ts @@ -0,0 +1,72 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { + $sessionControlBySession, + resetSessionControlForTests, + type SessionControlSnapshot +} from '@/store/session-control' + +import { handleControlEvent } from './session-control' +import type { GatewayEventContext } from './types' + +const SNAPSHOT: SessionControlSnapshot = { + goal: null, + heartbeat: null, + loop: null, + revision: 'event-revision', + updated_at: 1_700_000_100 +} + +function context(overrides: Partial = {}): GatewayEventContext { + return { + deps: {} as GatewayEventContext['deps'], + event: { payload: { control: SNAPSHOT }, session_id: 'event-session', type: 'session.control.update' }, + explicitSid: 'event-session', + fromActiveSource: () => true, + isActiveEvent: true, + occurredAt: 1_700_000_100, + payload: { control: SNAPSHOT } as GatewayEventContext['payload'], + scheduleConfigRefresh: vi.fn(), + sessionId: 'routed-session', + ...overrides + } +} + +describe('handleControlEvent', () => { + afterEach(() => { + resetSessionControlForTests() + }) + + it('does not claim non-control events', () => { + expect(handleControlEvent(context({ event: { type: 'message.complete' } }))).toBe(false) + }) + + it('applies a valid event to the routed session and proves it supported', () => { + expect(handleControlEvent(context())).toBe(true) + + expect($sessionControlBySession.get()).toMatchObject({ + 'routed-session': { capability: 'supported', snapshot: { revision: 'event-revision' } } + }) + expect($sessionControlBySession.get()['event-session']).toBeUndefined() + }) + + it('claims malformed control events without replacing the last good state', () => { + handleControlEvent(context()) + const first = $sessionControlBySession.get()['routed-session'] + + expect( + handleControlEvent( + context({ + payload: { control: { ...SNAPSHOT, updated_at: Number.POSITIVE_INFINITY } } as GatewayEventContext['payload'] + }) + ) + ).toBe(true) + + expect($sessionControlBySession.get()['routed-session']).toBe(first) + }) + + it('claims an unscoped control event without creating state', () => { + expect(handleControlEvent(context({ sessionId: null }))).toBe(true) + expect($sessionControlBySession.get()).toEqual({}) + }) +}) diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/session-control.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/session-control.ts new file mode 100644 index 0000000000..7f9f27e424 --- /dev/null +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/session-control.ts @@ -0,0 +1,20 @@ +import { applySessionControlUpdate } from '@/store/session-control' + +import type { GatewayEventContext } from './types' + +export function handleControlEvent(ctx: GatewayEventContext): boolean { + const { event, payload, sessionId } = ctx + + if (event.type !== 'session.control.update') { + return false + } + + if (!sessionId) { + return true + } + + const control = payload && typeof payload === 'object' ? (payload as { control?: unknown }).control : undefined + applySessionControlUpdate(sessionId, control) + + return true +} diff --git a/apps/desktop/src/store/session-control.test.ts b/apps/desktop/src/store/session-control.test.ts new file mode 100644 index 0000000000..c4cc9f1f72 --- /dev/null +++ b/apps/desktop/src/store/session-control.test.ts @@ -0,0 +1,457 @@ +import { JsonRpcGatewayError } from '@hermes/shared' +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +const { refreshLegacyGoal } = vi.hoisted(() => ({ refreshLegacyGoal: vi.fn() })) + +vi.mock('./goals', async importOriginal => ({ + ...(await importOriginal()), + refreshSessionGoal: refreshLegacyGoal +})) + +import { $gateway } from './gateway' +import { resetBackgroundPollingGuard } from './runtime-gone' +import { + $sessionControlBySession, + applySessionControlSnapshot, + applySessionControlUpdate, + clearSessionControl, + parseSessionControlSnapshot, + refreshSessionControl, + refreshSupportedSessionControlAfterTurn, + resetSessionControlAfterGatewayRebind, + resetSessionControlForTests, + runSessionControlAction, + type SessionControlSnapshot +} from './session-control' + +const FULL_SNAPSHOT: SessionControlSnapshot = { + goal: { + contract: { + boundaries: 'desktop store only', + constraints: 'do not lose state', + outcome: 'session control is hydrated', + stop_when: 'a human decision is required', + verification: 'focused tests pass' + }, + created_at: 1_700_000_000, + gates: [{ attempts: 0, command: 'npm test', last_exit_code: null, max_retries: 2, timeout_seconds: 60 }], + max_turns: 20, + status: 'active', + subgoals: ['write tests', 'repair state'], + title: 'Repair session control', + turns_used: 3, + updated_at: 1_700_000_100, + wait_barrier: { reason: 'waiting for deploy', type: 'until', until_at: 1_700_000_200 } + }, + heartbeat: { + created_at: 1_700_000_000, + fire_count: 5, + interval_seconds: 600, + last_fired_at: 1_700_000_100, + prompt: 'check health', + status: 'active' + }, + loop: { + awaiting_response: false, + created_at: 1_700_000_000, + current_delay: 300, + deferred_by_goal: false, + interval_seconds: 300, + last_fired_at: 1_700_000_100, + max_ticks: 10, + mode: 'interval', + next_due_at: 1_700_000_400, + prompt: 'check build status', + status: 'active', + ticks_fired: 3, + times: 3, + until: '' + }, + revision: 'revision-1', + updated_at: 1_700_000_100 +} + +function deferred() { + let resolve!: (value: T) => void + let reject!: (error: unknown) => void + + const promise = new Promise((resolvePromise, rejectPromise) => { + resolve = resolvePromise + reject = rejectPromise + }) + + return { promise, reject, resolve } +} + +function useGateway(request: (method: string, params: Record) => Promise): void { + $gateway.set({ request } as never) +} + +describe('session-control store', () => { + beforeEach(() => { + refreshLegacyGoal.mockReset() + resetSessionControlForTests() + }) + + afterEach(() => { + $gateway.set(null as never) + resetBackgroundPollingGuard() + resetSessionControlForTests() + }) + + it('parses the exact persisted goal, loop, heartbeat, and wait-barrier shapes into fresh data', () => { + const parsed = parseSessionControlSnapshot(FULL_SNAPSHOT) + + expect(parsed).toEqual(FULL_SNAPSHOT) + expect(parsed).not.toBe(FULL_SNAPSHOT) + expect(parsed!.goal).not.toBe(FULL_SNAPSHOT.goal) + expect(parsed!.goal!.contract).not.toBe(FULL_SNAPSHOT.goal!.contract) + expect(parsed!.loop!.mode).toBe('interval') + expect(parsed!.goal!.wait_barrier).toEqual({ reason: 'waiting for deploy', type: 'until', until_at: 1_700_000_200 }) + }) + + it.each([ + ['unknown goal status', { ...FULL_SNAPSHOT, goal: { ...FULL_SNAPSHOT.goal!, status: 'waiting' } }], + ['non-finite top-level timestamp', { ...FULL_SNAPSHOT, updated_at: Number.NaN }], + ['malformed goal contract', { ...FULL_SNAPSHOT, goal: { ...FULL_SNAPSHOT.goal!, contract: { outcome: 3 } } }], + [ + 'gate output that is not in the allowlisted summary', + { + ...FULL_SNAPSHOT, + goal: { ...FULL_SNAPSHOT.goal!, gates: [{ ...FULL_SNAPSHOT.goal!.gates[0], last_output_tail: 'leak' }] } + } + ], + [ + 'wait target that does not match its discriminator', + { ...FULL_SNAPSHOT, goal: { ...FULL_SNAPSHOT.goal!, wait_barrier: { reason: 'pid', target: '7', type: 'pid' } } } + ], + ['unknown loop mode', { ...FULL_SNAPSHOT, loop: { ...FULL_SNAPSHOT.loop!, mode: 'fixed' } }], + ['malformed heartbeat', { ...FULL_SNAPSHOT, heartbeat: { ...FULL_SNAPSHOT.heartbeat!, fire_count: 'five' } }], + ['unknown top-level field', { ...FULL_SNAPSHOT, unsupported: true }] + ])('rejects %s without accepting a partial snapshot', (_name, value) => { + expect(parseSessionControlSnapshot(value)).toBeNull() + }) + + it('preserves entry and snapshot identity for a same-revision snapshot with no other state change', () => { + applySessionControlSnapshot('s1', FULL_SNAPSHOT) + const first = $sessionControlBySession.get().s1 + + applySessionControlSnapshot('s1', { ...FULL_SNAPSHOT }) + const second = $sessionControlBySession.get().s1 + + expect(second).toBe(first) + expect(second!.snapshot).toBe(first!.snapshot) + }) + + it('creates foreground loading state and leaves background hydration visually quiet', async () => { + const first = deferred() + const request = vi.fn(() => first.promise) + useGateway(request) + + const foreground = refreshSessionControl('s1') + expect($sessionControlBySession.get().s1).toMatchObject({ + capability: 'unknown', + error: null, + loading: true, + pendingAction: null + }) + first.resolve({ control: FULL_SNAPSHOT }) + await foreground + + const second = deferred() + request.mockImplementationOnce(() => second.promise) + const background = refreshSessionControl('s1', { background: true }) + + expect($sessionControlBySession.get().s1).toMatchObject({ capability: 'supported', loading: false }) + second.resolve({ control: { ...FULL_SNAPSHOT, revision: 'background-revision' } }) + await background + }) + + it('falls back once on the unsupported transition and suppresses future compatibility retries', async () => { + const request = vi.fn(async () => { + throw new JsonRpcGatewayError('method not found', { code: -32601 }) + }) + + useGateway(request) + + await refreshSessionControl('s1') + await refreshSessionControl('s1') + + expect($sessionControlBySession.get().s1).toMatchObject({ + capability: 'unsupported', + loading: false, + snapshot: null + }) + expect(refreshLegacyGoal).toHaveBeenCalledTimes(1) + expect(refreshLegacyGoal).toHaveBeenCalledWith('s1') + expect(request).toHaveBeenCalledTimes(1) + }) + + it('lets a precise gateway-rebind seam make an unsupported session probeable again', async () => { + useGateway( + vi.fn(async () => { + throw new JsonRpcGatewayError('method not found', { code: -32601 }) + }) + ) + await refreshSessionControl('s1') + + resetSessionControlAfterGatewayRebind() + expect($sessionControlBySession.get().s1).toMatchObject({ capability: 'unknown', loading: false }) + + const request = vi.fn(async () => ({ control: FULL_SNAPSHOT })) + useGateway(request) + await refreshSessionControl('s1') + expect(request).toHaveBeenCalledTimes(1) + expect($sessionControlBySession.get().s1!.capability).toBe('supported') + }) + + it('retains the last good snapshot and reports a bounded ordinary read error', async () => { + applySessionControlSnapshot('s1', FULL_SNAPSHOT) + const message = 'x'.repeat(500) + useGateway( + vi.fn(async () => { + throw new Error(message) + }) + ) + + await expect(refreshSessionControl('s1', { background: true })).resolves.toBeDefined() + + const entry = $sessionControlBySession.get().s1! + expect(entry.snapshot!.revision).toBe(FULL_SNAPSHOT.revision) + expect(entry.capability).toBe('supported') + expect(entry.error).toHaveLength(240) + expect(entry.loading).toBe(false) + }) + + it('marks a session gone without publishing an ordinary hydration error', async () => { + useGateway( + vi.fn(async () => { + throw new JsonRpcGatewayError('session not found', { code: 4001 }) + }) + ) + + await refreshSessionControl('gone') + expect($sessionControlBySession.get().gone).toMatchObject({ error: null, loading: false }) + }) + + it('rejects a session-gone action truthfully while clearing only its pending state', async () => { + applySessionControlSnapshot('s1', FULL_SNAPSHOT) + useGateway( + vi.fn(async () => { + throw new JsonRpcGatewayError('session not found', { code: 4001 }) + }) + ) + + await expect(runSessionControlAction('s1', 'goal.pause')).rejects.toThrow('session not found') + expect($sessionControlBySession.get().s1).toMatchObject({ error: null, pendingAction: null }) + expect($sessionControlBySession.get().s1!.snapshot!.revision).toBe(FULL_SNAPSHOT.revision) + }) + + it('makes an event-first session supported and leaves malformed events as a no-op', () => { + applySessionControlUpdate('s1', FULL_SNAPSHOT) + const first = $sessionControlBySession.get().s1 + + applySessionControlUpdate('s1', { ...FULL_SNAPSHOT, loop: { ...FULL_SNAPSHOT.loop!, mode: 'fixed' } }) + + expect(first).toMatchObject({ capability: 'supported', pendingAction: null }) + expect($sessionControlBySession.get().s1).toBe(first) + }) + + it('submits only the requested action and exposes its pending state on that session', async () => { + const response = deferred() + const request = vi.fn(() => response.promise) + useGateway(request) + + const action = runSessionControlAction('s1', 'subgoal.add', { text: 'verify hydration' }) + expect($sessionControlBySession.get().s1).toMatchObject({ + capability: 'unknown', + loading: false, + pendingAction: 'subgoal.add' + }) + expect($sessionControlBySession.get().s2).toBeUndefined() + expect(request).toHaveBeenCalledWith('session.control', { + action: 'subgoal.add', + args: { text: 'verify hydration' }, + session_id: 's1' + }) + + response.resolve({ + control: { ...FULL_SNAPSHOT, revision: 'action-revision' }, + dispatch: { display: null, message: null, notice: null, output: 'added', type: 'exec' } + }) + + await expect(action).resolves.toEqual({ display: null, message: null, notice: null, output: 'added', type: 'exec' }) + expect($sessionControlBySession.get().s1).toMatchObject({ + capability: 'supported', + error: null, + pendingAction: null + }) + }) + + it('rejects a failed action truthfully while retaining the good snapshot and clearing its pending state', async () => { + applySessionControlSnapshot('s1', FULL_SNAPSHOT) + useGateway( + vi.fn(async () => { + throw new Error('backend unavailable') + }) + ) + + await expect(runSessionControlAction('s1', 'goal.pause')).rejects.toThrow('backend unavailable') + + expect($sessionControlBySession.get().s1).toMatchObject({ error: 'backend unavailable', pendingAction: null }) + expect($sessionControlBySession.get().s1!.snapshot!.revision).toBe(FULL_SNAPSHOT.revision) + }) + + it('rejects malformed action data without promoting an unknown capability', async () => { + useGateway( + vi.fn(async () => ({ + control: FULL_SNAPSHOT, + dispatch: { display: null, message: null, notice: null, output: null, type: 'unknown' } + })) + ) + + await expect(runSessionControlAction('s1', 'goal.pause')).rejects.toThrow('Invalid session.control action response') + expect($sessionControlBySession.get().s1).toMatchObject({ + capability: 'unknown', + pendingAction: null, + snapshot: null + }) + }) + + it('does not let a late read overwrite a newer event', async () => { + const slow = deferred() + useGateway(vi.fn(() => slow.promise)) + + const read = refreshSessionControl('s1') + applySessionControlUpdate('s1', { ...FULL_SNAPSHOT, revision: 'event-newer' }) + slow.resolve({ control: { ...FULL_SNAPSHOT, revision: 'read-stale' } }) + await read + + expect($sessionControlBySession.get().s1!.snapshot!.revision).toBe('event-newer') + }) + + it('does not let a late read overwrite a newer action response', async () => { + const slow = deferred() + + const request = vi + .fn() + .mockImplementationOnce(() => slow.promise) + .mockResolvedValueOnce({ + control: { ...FULL_SNAPSHOT, revision: 'action-newer' }, + dispatch: { display: null, message: null, notice: null, output: 'paused', type: 'exec' } + }) + + useGateway(request) + + const read = refreshSessionControl('s1') + await runSessionControlAction('s1', 'goal.pause') + slow.resolve({ control: { ...FULL_SNAPSHOT, revision: 'read-stale' } }) + await read + + expect($sessionControlBySession.get().s1!.snapshot!.revision).toBe('action-newer') + }) + + it('does not let a late read repopulate a cleared session', async () => { + const slow = deferred() + useGateway(vi.fn(() => slow.promise)) + + const read = refreshSessionControl('s1') + clearSessionControl('s1') + slow.resolve({ control: FULL_SNAPSHOT }) + await read + + expect($sessionControlBySession.get().s1).toBeUndefined() + }) + + it('does not let an older read overwrite a newer read', async () => { + const slow = deferred() + const fast = deferred() + + const request = vi + .fn() + .mockImplementationOnce(() => slow.promise) + .mockImplementationOnce(() => fast.promise) + + useGateway(request) + + const older = refreshSessionControl('s1') + const newer = refreshSessionControl('s1') + fast.resolve({ control: { ...FULL_SNAPSHOT, revision: 'newer-read' } }) + await newer + slow.resolve({ control: { ...FULL_SNAPSHOT, revision: 'older-read' } }) + await older + + expect($sessionControlBySession.get().s1!.snapshot!.revision).toBe('newer-read') + }) + + it('does not let a late post-turn refresh overwrite a newer event', async () => { + applySessionControlSnapshot('s1', FULL_SNAPSHOT) + const slow = deferred() + useGateway(vi.fn(() => slow.promise)) + + const refresh = refreshSupportedSessionControlAfterTurn('s1') + applySessionControlUpdate('s1', { ...FULL_SNAPSHOT, revision: 'event-newer' }) + slow.resolve({ control: { ...FULL_SNAPSHOT, revision: 'post-turn-stale' } }) + await refresh + + expect($sessionControlBySession.get().s1!.snapshot!.revision).toBe('event-newer') + }) + + it('skips post-turn refreshes until the session is known to support the control RPC', async () => { + const request = vi.fn(async () => ({ control: FULL_SNAPSHOT })) + useGateway(request) + + await refreshSupportedSessionControlAfterTurn('unknown') + expect(request).not.toHaveBeenCalled() + + applySessionControlUpdate('unsupported', FULL_SNAPSHOT) + useGateway( + vi.fn(async () => { + throw new JsonRpcGatewayError('method not found', { code: -32601 }) + }) + ) + await refreshSessionControl('unsupported') + + const unsupportedRequest = vi.fn(async () => ({ control: FULL_SNAPSHOT })) + useGateway(unsupportedRequest) + await refreshSupportedSessionControlAfterTurn('unsupported') + expect(unsupportedRequest).not.toHaveBeenCalled() + }) + + it('returns a valid stale action dispatch without letting it overwrite a newer event', async () => { + const slow = deferred() + useGateway(vi.fn(() => slow.promise)) + + const action = runSessionControlAction('s1', 'goal.pause') + applySessionControlUpdate('s1', { ...FULL_SNAPSHOT, revision: 'event-newer' }) + slow.resolve({ + control: { ...FULL_SNAPSHOT, revision: 'action-stale' }, + dispatch: { display: null, message: null, notice: null, output: 'paused', type: 'exec' } + }) + + await expect(action).resolves.toMatchObject({ output: 'paused', type: 'exec' }) + expect($sessionControlBySession.get().s1!.snapshot!.revision).toBe('event-newer') + }) + + it('does not schedule a timer or poller while hydrating or dispatching control state', async () => { + const setIntervalSpy = vi.spyOn(globalThis, 'setInterval') + const setTimeoutSpy = vi.spyOn(globalThis, 'setTimeout') + useGateway( + vi.fn(async method => + method === 'session.control.read' + ? { control: FULL_SNAPSHOT } + : { + control: FULL_SNAPSHOT, + dispatch: { display: null, message: null, notice: null, output: null, type: 'exec' } + } + ) + ) + + await refreshSessionControl('s1') + await runSessionControlAction('s1', 'goal.pause') + await refreshSupportedSessionControlAfterTurn('s1') + + expect(setIntervalSpy).not.toHaveBeenCalled() + expect(setTimeoutSpy).not.toHaveBeenCalled() + }) +}) diff --git a/apps/desktop/src/store/session-control.ts b/apps/desktop/src/store/session-control.ts new file mode 100644 index 0000000000..9e47430090 --- /dev/null +++ b/apps/desktop/src/store/session-control.ts @@ -0,0 +1,838 @@ +import { atom } from 'nanostores' + +import { $gateway } from './gateway' +import { refreshSessionGoal } from './goals' +import { isSessionGone, isSessionGoneForBackgroundPolling, markSessionGone } from './runtime-gone' +import { ambientRequestFor } from './session-gone-latch' +import { requestForOwnedSession } from './session-states' + +export type SessionControlGoalStatus = 'active' | 'done' | 'paused' +export type SessionControlLoopMode = 'interval' | 'self_paced' +export type SessionControlLoopStatus = 'active' | 'done' | 'paused' +export type SessionControlHeartbeatStatus = 'active' | 'paused' + +export interface SessionControlGoalContract { + boundaries: string + constraints: string + outcome: string + stop_when: string + verification: string +} + +export interface SessionControlGate { + attempts: number + command: string + last_exit_code: number | null + max_retries: number + timeout_seconds: number +} + +export type SessionControlWaitBarrier = + | { reason: string; type: 'until'; until_at: number } + | { reason: string; target: string; type: 'session' } + | { reason: string; target: number; type: 'pid' } + +export interface SessionControlGoal { + contract: SessionControlGoalContract + gates: SessionControlGate[] + last_reason?: string + last_verdict?: 'blocked' | 'continue' | 'done' | 'skipped' | 'wait' + max_turns: number + paused_reason?: string + status: SessionControlGoalStatus + subgoals: string[] + title: string + turns_used: number + updated_at?: number + wait_barrier?: SessionControlWaitBarrier + created_at?: number +} + +export interface SessionControlLoop { + awaiting_response: boolean + created_at: number + current_delay: number + deferred_by_goal: boolean + interval_seconds: number + last_fired_at: number + last_stop_reason?: string + max_ticks: number + mode: SessionControlLoopMode + next_due_at: number + paused_reason?: string + prompt: string + status: SessionControlLoopStatus + ticks_fired: number + times: number + until: string +} + +export interface SessionControlHeartbeat { + created_at: number + fire_count: number + interval_seconds: number + last_fired_at: number + prompt: string + status: SessionControlHeartbeatStatus +} + +export interface SessionControlSnapshot { + goal: SessionControlGoal | null + heartbeat: SessionControlHeartbeat | null + loop: SessionControlLoop | null + revision: string + updated_at: number +} + +export type SessionControlAction = + | 'goal.clear' + | 'goal.pause' + | 'goal.resume' + | 'goal.unwait' + | 'heartbeat.clear' + | 'heartbeat.pause' + | 'heartbeat.resume' + | 'loop.pause' + | 'loop.resume' + | 'loop.stop' + | 'subgoal.add' + | 'subgoal.clear' + | 'subgoal.remove' + +export type SessionControlActionArgs = { index: number } | { text: string } + +export interface SessionControlDispatch { + display: string | null + message: string | null + notice: string | null + output: string | null + type: 'exec' | 'send' +} + +export interface SessionControlEntry { + capability: 'unknown' | 'supported' | 'unsupported' + error: string | null + loading: boolean + pendingAction: SessionControlAction | null + snapshot: SessionControlSnapshot | null +} + +interface RefreshOptions { + background?: boolean +} + +type UnknownRecord = Record + +const GOAL_STATUSES = new Set(['active', 'done', 'paused']) + +const GOAL_VERDICTS = new Set>([ + 'blocked', + 'continue', + 'done', + 'skipped', + 'wait' +]) + +const LOOP_MODES = new Set(['interval', 'self_paced']) +const LOOP_STATUSES = new Set(['active', 'done', 'paused']) +const HEARTBEAT_STATUSES = new Set(['active', 'paused']) +const DISPATCH_TYPES = new Set(['exec', 'send']) +const ERROR_LIMIT = 240 + +const versions = new Map() + +export const $sessionControlBySession = atom>({}) + +function isRecord(value: unknown): value is UnknownRecord { + return value !== null && typeof value === 'object' && !Array.isArray(value) +} + +function hasOwn(value: UnknownRecord, key: string): boolean { + return Object.prototype.hasOwnProperty.call(value, key) +} + +function hasExactFields(value: UnknownRecord, required: string[], optional: string[] = []): boolean { + const allowed = new Set([...required, ...optional]) + + return required.every(key => hasOwn(value, key)) && Object.keys(value).every(key => allowed.has(key)) +} + +function isFiniteNumber(value: unknown): value is number { + return typeof value === 'number' && Number.isFinite(value) +} + +function isInteger(value: unknown): value is number { + return Number.isInteger(value) +} + +function hasOptionalStrings(value: UnknownRecord, keys: string[]): boolean { + return keys.every(key => !hasOwn(value, key) || typeof value[key] === 'string') +} + +function parseGoalContract(value: unknown): SessionControlGoalContract | null { + if ( + !isRecord(value) || + !hasExactFields(value, ['outcome', 'verification', 'constraints', 'boundaries', 'stop_when']) + ) { + return null + } + + if ( + typeof value.outcome !== 'string' || + typeof value.verification !== 'string' || + typeof value.constraints !== 'string' || + typeof value.boundaries !== 'string' || + typeof value.stop_when !== 'string' + ) { + return null + } + + return { + boundaries: value.boundaries, + constraints: value.constraints, + outcome: value.outcome, + stop_when: value.stop_when, + verification: value.verification + } +} + +function parseGate(value: unknown): SessionControlGate | null { + if ( + !isRecord(value) || + !hasExactFields(value, ['command', 'timeout_seconds', 'max_retries', 'attempts', 'last_exit_code']) + ) { + return null + } + + if ( + typeof value.command !== 'string' || + !isInteger(value.timeout_seconds) || + !isInteger(value.max_retries) || + !isInteger(value.attempts) || + (value.last_exit_code !== null && !isInteger(value.last_exit_code)) + ) { + return null + } + + return { + attempts: value.attempts, + command: value.command, + last_exit_code: value.last_exit_code, + max_retries: value.max_retries, + timeout_seconds: value.timeout_seconds + } +} + +function parseWaitBarrier(value: unknown): SessionControlWaitBarrier | null { + if (!isRecord(value) || typeof value.type !== 'string') { + return null + } + + if (value.type === 'until') { + if ( + !hasExactFields(value, ['type', 'until_at', 'reason']) || + !isFiniteNumber(value.until_at) || + typeof value.reason !== 'string' + ) { + return null + } + + return { reason: value.reason, type: 'until', until_at: value.until_at } + } + + if (value.type === 'session') { + if ( + !hasExactFields(value, ['type', 'target', 'reason']) || + typeof value.target !== 'string' || + typeof value.reason !== 'string' + ) { + return null + } + + return { reason: value.reason, target: value.target, type: 'session' } + } + + if (value.type === 'pid') { + if ( + !hasExactFields(value, ['type', 'target', 'reason']) || + !isInteger(value.target) || + typeof value.reason !== 'string' + ) { + return null + } + + return { reason: value.reason, target: value.target, type: 'pid' } + } + + return null +} + +function parseGoal(value: unknown): SessionControlGoal | null { + const required = ['title', 'status', 'turns_used', 'max_turns', 'contract', 'subgoals', 'gates'] + const optional = ['created_at', 'updated_at', 'paused_reason', 'last_verdict', 'last_reason', 'wait_barrier'] + + if (!isRecord(value) || !hasExactFields(value, required, optional)) { + return null + } + + if ( + typeof value.title !== 'string' || + typeof value.status !== 'string' || + !GOAL_STATUSES.has(value.status as SessionControlGoalStatus) || + !isInteger(value.turns_used) || + !isInteger(value.max_turns) || + !Array.isArray(value.subgoals) || + !value.subgoals.every(subgoal => typeof subgoal === 'string') || + !Array.isArray(value.gates) || + !hasOptionalStrings(value, ['paused_reason', 'last_reason']) || + (hasOwn(value, 'created_at') && !isFiniteNumber(value.created_at)) || + (hasOwn(value, 'updated_at') && !isFiniteNumber(value.updated_at)) || + (hasOwn(value, 'last_verdict') && + (typeof value.last_verdict !== 'string' || + !GOAL_VERDICTS.has(value.last_verdict as NonNullable))) + ) { + return null + } + + const contract = parseGoalContract(value.contract) + const parsedGates = value.gates.map(parseGate) + const waitBarrier = hasOwn(value, 'wait_barrier') ? parseWaitBarrier(value.wait_barrier) : undefined + + if (!contract || parsedGates.some(gate => gate === null) || (hasOwn(value, 'wait_barrier') && !waitBarrier)) { + return null + } + + const gates = parsedGates.filter((gate): gate is SessionControlGate => gate !== null) + + const goal: SessionControlGoal = { + contract, + gates, + max_turns: value.max_turns, + status: value.status as SessionControlGoalStatus, + subgoals: [...value.subgoals], + title: value.title, + turns_used: value.turns_used + } + + if (hasOwn(value, 'created_at')) { + goal.created_at = value.created_at as number + } + + if (hasOwn(value, 'updated_at')) { + goal.updated_at = value.updated_at as number + } + + if (hasOwn(value, 'paused_reason')) { + goal.paused_reason = value.paused_reason as string + } + + if (hasOwn(value, 'last_reason')) { + goal.last_reason = value.last_reason as string + } + + if (hasOwn(value, 'last_verdict')) { + goal.last_verdict = value.last_verdict as SessionControlGoal['last_verdict'] + } + + if (waitBarrier) { + goal.wait_barrier = waitBarrier + } + + return goal +} + +function parseLoop(value: unknown): SessionControlLoop | null { + const required = [ + 'prompt', + 'status', + 'mode', + 'interval_seconds', + 'current_delay', + 'times', + 'until', + 'max_ticks', + 'ticks_fired', + 'created_at', + 'last_fired_at', + 'next_due_at', + 'awaiting_response', + 'deferred_by_goal' + ] + + if (!isRecord(value) || !hasExactFields(value, required, ['paused_reason', 'last_stop_reason'])) { + return null + } + + if ( + typeof value.prompt !== 'string' || + typeof value.status !== 'string' || + !LOOP_STATUSES.has(value.status as SessionControlLoopStatus) || + typeof value.mode !== 'string' || + !LOOP_MODES.has(value.mode as SessionControlLoopMode) || + !isFiniteNumber(value.interval_seconds) || + !isFiniteNumber(value.current_delay) || + !isInteger(value.times) || + typeof value.until !== 'string' || + !isInteger(value.max_ticks) || + !isInteger(value.ticks_fired) || + !isFiniteNumber(value.created_at) || + !isFiniteNumber(value.last_fired_at) || + !isFiniteNumber(value.next_due_at) || + typeof value.awaiting_response !== 'boolean' || + typeof value.deferred_by_goal !== 'boolean' || + !hasOptionalStrings(value, ['paused_reason', 'last_stop_reason']) + ) { + return null + } + + const loop: SessionControlLoop = { + awaiting_response: value.awaiting_response, + created_at: value.created_at, + current_delay: value.current_delay, + deferred_by_goal: value.deferred_by_goal, + interval_seconds: value.interval_seconds, + last_fired_at: value.last_fired_at, + max_ticks: value.max_ticks, + mode: value.mode as SessionControlLoopMode, + next_due_at: value.next_due_at, + prompt: value.prompt, + status: value.status as SessionControlLoopStatus, + ticks_fired: value.ticks_fired, + times: value.times, + until: value.until + } + + if (hasOwn(value, 'paused_reason')) { + loop.paused_reason = value.paused_reason as string + } + + if (hasOwn(value, 'last_stop_reason')) { + loop.last_stop_reason = value.last_stop_reason as string + } + + return loop +} + +function parseHeartbeat(value: unknown): SessionControlHeartbeat | null { + const required = ['prompt', 'status', 'interval_seconds', 'created_at', 'last_fired_at', 'fire_count'] + + if (!isRecord(value) || !hasExactFields(value, required)) { + return null + } + + if ( + typeof value.prompt !== 'string' || + typeof value.status !== 'string' || + !HEARTBEAT_STATUSES.has(value.status as SessionControlHeartbeatStatus) || + !isInteger(value.interval_seconds) || + !isFiniteNumber(value.created_at) || + !isFiniteNumber(value.last_fired_at) || + !isInteger(value.fire_count) + ) { + return null + } + + return { + created_at: value.created_at, + fire_count: value.fire_count, + interval_seconds: value.interval_seconds, + last_fired_at: value.last_fired_at, + prompt: value.prompt, + status: value.status as SessionControlHeartbeatStatus + } +} + +/** Parses the stable allowlisted backend shape into fresh renderer-owned data. */ +export function parseSessionControlSnapshot(value: unknown): SessionControlSnapshot | null { + if (!isRecord(value) || !hasExactFields(value, ['goal', 'loop', 'heartbeat', 'revision', 'updated_at'])) { + return null + } + + if (typeof value.revision !== 'string' || !isFiniteNumber(value.updated_at)) { + return null + } + + const goal = value.goal === null ? null : parseGoal(value.goal) + const loop = value.loop === null ? null : parseLoop(value.loop) + const heartbeat = value.heartbeat === null ? null : parseHeartbeat(value.heartbeat) + + if ((value.goal !== null && !goal) || (value.loop !== null && !loop) || (value.heartbeat !== null && !heartbeat)) { + return null + } + + return { goal, heartbeat, loop, revision: value.revision, updated_at: value.updated_at } +} + +function parseSessionControlDispatch(value: unknown): SessionControlDispatch | null { + if (!isRecord(value) || !hasExactFields(value, ['type', 'output', 'notice', 'message', 'display'])) { + return null + } + + if ( + typeof value.type !== 'string' || + !DISPATCH_TYPES.has(value.type as SessionControlDispatch['type']) || + ![value.output, value.notice, value.message, value.display].every( + field => field === null || typeof field === 'string' + ) + ) { + return null + } + + return { + display: value.display as string | null, + message: value.message as string | null, + notice: value.notice as string | null, + output: value.output as string | null, + type: value.type as SessionControlDispatch['type'] + } +} + +function emptyEntry(): SessionControlEntry { + return { capability: 'unknown', error: null, loading: false, pendingAction: null, snapshot: null } +} + +function currentVersion(sessionId: string): number { + return versions.get(sessionId) ?? 0 +} + +function advanceVersion(sessionId: string): number { + const next = currentVersion(sessionId) + 1 + versions.set(sessionId, next) + + return next +} + +function isCurrent(sessionId: string, token: number): boolean { + return currentVersion(sessionId) === token +} + +function sameEntry(first: SessionControlEntry, second: SessionControlEntry): boolean { + return ( + first.capability === second.capability && + first.error === second.error && + first.loading === second.loading && + first.pendingAction === second.pendingAction && + first.snapshot === second.snapshot + ) +} + +function publishEntry(sessionId: string, next: SessionControlEntry): SessionControlEntry { + const entries = $sessionControlBySession.get() + const current = entries[sessionId] + + if (current && sameEntry(current, next)) { + return current + } + + $sessionControlBySession.set({ ...entries, [sessionId]: next }) + + return next +} + +function applyParsedSnapshot(sessionId: string, snapshot: SessionControlSnapshot): SessionControlEntry { + advanceVersion(sessionId) + const current = $sessionControlBySession.get()[sessionId] ?? emptyEntry() + const nextSnapshot = current.snapshot?.revision === snapshot.revision ? current.snapshot : snapshot + + return publishEntry(sessionId, { + capability: 'supported', + error: null, + loading: false, + pendingAction: null, + snapshot: nextSnapshot + }) +} + +/** Applies a valid read/action snapshot and marks its session as supported. */ +export function applySessionControlSnapshot(sessionId: string, rawSnapshot: unknown): SessionControlEntry | undefined { + if (!sessionId) { + return undefined + } + + const snapshot = parseSessionControlSnapshot(rawSnapshot) + + return snapshot ? applyParsedSnapshot(sessionId, snapshot) : undefined +} + +/** Applies an event update; invalid updates are deliberately a claimed no-op. */ +export function applySessionControlUpdate(sessionId: string, rawSnapshot: unknown): SessionControlEntry | undefined { + return applySessionControlSnapshot(sessionId, rawSnapshot) +} + +export function clearSessionControl(sessionId: string): void { + if (!sessionId) { + return + } + + advanceVersion(sessionId) + const entries = $sessionControlBySession.get() + + if (!(sessionId in entries)) { + return + } + + const { [sessionId]: _removed, ...remaining } = entries + $sessionControlBySession.set(remaining) +} + +/** Lets an explicitly rebound gateway probe backends that were unsupported on its predecessor. */ +export function resetSessionControlAfterGatewayRebind(): void { + const entries = $sessionControlBySession.get() + let changed = false + const nextEntries: Record = {} + + for (const [sessionId, entry] of Object.entries(entries)) { + advanceVersion(sessionId) + + if (entry.capability === 'unsupported') { + nextEntries[sessionId] = { ...entry, capability: 'unknown', error: null, loading: false, pendingAction: null } + changed = true + } else { + nextEntries[sessionId] = entry + } + } + + if (changed) { + $sessionControlBySession.set(nextEntries) + } +} + +/** Narrow reset seam for tests; production reconnect behavior remains explicitly owned by its caller. */ +export function resetSessionControlForTests(): void { + for (const sessionId of new Set([...versions.keys(), ...Object.keys($sessionControlBySession.get())])) { + advanceVersion(sessionId) + } + + $sessionControlBySession.set({}) + versions.clear() +} + +function beginRead(sessionId: string, background: boolean): number { + const token = advanceVersion(sessionId) + const current = $sessionControlBySession.get()[sessionId] ?? emptyEntry() + + publishEntry(sessionId, { + ...current, + error: null, + loading: background ? current.loading : true + }) + + return token +} + +function beginAction(sessionId: string, action: SessionControlAction): number { + const token = advanceVersion(sessionId) + const current = $sessionControlBySession.get()[sessionId] ?? emptyEntry() + + publishEntry(sessionId, { + ...current, + error: null, + loading: false, + pendingAction: action + }) + + return token +} + +function boundedError(error: unknown): string { + const message = + error instanceof Error + ? error.message + : isRecord(error) && typeof error.message === 'string' + ? error.message + : 'Session control request failed' + + return message.trim().slice(0, ERROR_LIMIT) || 'Session control request failed' +} + +function publishFailure(sessionId: string, token: number, error: unknown, clearPendingAction: boolean): void { + if (!isCurrent(sessionId, token)) { + return + } + + const current = $sessionControlBySession.get()[sessionId] ?? emptyEntry() + publishEntry(sessionId, { + ...current, + error: boundedError(error), + loading: false, + pendingAction: clearPendingAction ? null : current.pendingAction + }) +} + +function finishGoneRequest(sessionId: string, token: number, clearPendingAction: boolean): void { + if (!isCurrent(sessionId, token)) { + return + } + + markSessionGone(sessionId) + const current = $sessionControlBySession.get()[sessionId] ?? emptyEntry() + publishEntry(sessionId, { + ...current, + loading: false, + pendingAction: clearPendingAction ? null : current.pendingAction + }) +} + +function markUnsupported(sessionId: string, token: number): boolean { + if (!isCurrent(sessionId, token)) { + return false + } + + const current = $sessionControlBySession.get()[sessionId] ?? emptyEntry() + + if (current.capability === 'unsupported') { + return false + } + + advanceVersion(sessionId) + publishEntry(sessionId, { + ...current, + capability: 'unsupported', + error: null, + loading: false, + pendingAction: null + }) + + return true +} + +function isMethodNotFound(error: unknown): boolean { + if (isRecord(error) && error.code === -32601) { + return true + } + + const message = + error instanceof Error ? error.message : isRecord(error) && typeof error.message === 'string' ? error.message : '' + + return message.toLowerCase().includes('method not found') || message.toLowerCase().includes('method-not-found') +} + +/** Hydrates one session's structured controls; background refreshes never flash a loading state. */ +export async function refreshSessionControl( + sessionId: string, + options: RefreshOptions = {} +): Promise { + const existing = $sessionControlBySession.get()[sessionId] + + if (!sessionId || existing?.capability === 'unsupported' || isSessionGone(sessionId)) { + return existing + } + + const gateway = $gateway.get() + + if (!gateway) { + return existing + } + + const token = beginRead(sessionId, Boolean(options.background)) + + try { + const response = await requestForOwnedSession( + sessionId, + ambientRequestFor(gateway), + 'session.control.read', + { session_id: sessionId } + ) + + if (!isCurrent(sessionId, token)) { + return $sessionControlBySession.get()[sessionId] + } + + const snapshot = isRecord(response) ? parseSessionControlSnapshot(response.control) : null + + if (!snapshot) { + publishFailure(sessionId, token, new Error('Invalid session.control.read response'), false) + + return $sessionControlBySession.get()[sessionId] + } + + return applyParsedSnapshot(sessionId, snapshot) + } catch (error) { + if (!isCurrent(sessionId, token)) { + return $sessionControlBySession.get()[sessionId] + } + + if (isMethodNotFound(error)) { + const transitioned = markUnsupported(sessionId, token) + + if (transitioned) { + await refreshSessionGoal(sessionId) + } + + return $sessionControlBySession.get()[sessionId] + } + + if (isSessionGoneForBackgroundPolling(error)) { + finishGoneRequest(sessionId, token, false) + + return $sessionControlBySession.get()[sessionId] + } + + publishFailure(sessionId, token, error, false) + + return $sessionControlBySession.get()[sessionId] + } +} + +/** Runs an allowlisted backend control action; callers own any composer/UI dispatch. */ +export async function runSessionControlAction( + sessionId: string, + action: SessionControlAction, + args?: SessionControlActionArgs +): Promise { + if (!sessionId) { + throw new Error('A session id is required for session control') + } + + if (isSessionGone(sessionId)) { + throw new Error('Session not found') + } + + const gateway = $gateway.get() + + if (!gateway) { + throw new Error('Session control gateway is unavailable') + } + + const token = beginAction(sessionId, action) + + try { + const response = await requestForOwnedSession(sessionId, ambientRequestFor(gateway), 'session.control', { + action, + args: args ?? {}, + session_id: sessionId + }) + + const snapshot = isRecord(response) ? parseSessionControlSnapshot(response.control) : null + const dispatch = isRecord(response) ? parseSessionControlDispatch(response.dispatch) : null + + if (!snapshot || !dispatch) { + const error = new Error('Invalid session.control action response') + publishFailure(sessionId, token, error, true) + throw error + } + + if (isCurrent(sessionId, token)) { + applyParsedSnapshot(sessionId, snapshot) + } + + return dispatch + } catch (error) { + if (isSessionGoneForBackgroundPolling(error)) { + finishGoneRequest(sessionId, token, true) + } else { + publishFailure(sessionId, token, error, true) + } + + throw error + } +} + +/** Refreshes only sessions already proven to support the structured-control RPC. */ +export async function refreshSupportedSessionControlAfterTurn(sessionId: string): Promise { + if (!sessionId || $sessionControlBySession.get()[sessionId]?.capability !== 'supported') { + return + } + + await refreshSessionControl(sessionId, { background: true }) +} From dffd8d62c2857510e7f3fcdc7bb8290741b28791 Mon Sep 17 00:00:00 2001 From: Jerry Gooch Date: Sat, 5 Sep 2026 15:22:49 -0500 Subject: [PATCH 181/276] feat(desktop): add session automation controls (cherry picked from commit 8a61c2bcba8e4c7b3adcc172d4e5457721f76990) --- .../hooks/use-status-presence.test.ts | 200 +++ .../composer/hooks/use-status-presence.ts | 22 +- apps/desktop/src/app/chat/composer/index.tsx | 1 + .../app/chat/composer/status-stack/index.tsx | 39 +- .../status-stack/session-control-goal.tsx | 606 +++++++++ .../session-control-heartbeat.tsx | 195 +++ .../status-stack/session-control-loop.tsx | 239 ++++ .../status-stack/session-control-utils.ts | 82 ++ .../status-stack/session-control.test.tsx | 1146 +++++++++++++++++ .../composer/status-stack/session-control.tsx | 108 ++ apps/desktop/src/i18n/ar.ts | 85 ++ apps/desktop/src/i18n/en.ts | 85 ++ apps/desktop/src/i18n/ja.ts | 85 ++ apps/desktop/src/i18n/ru.ts | 85 ++ apps/desktop/src/i18n/types.ts | 85 ++ apps/desktop/src/i18n/zh-hant.ts | 85 ++ apps/desktop/src/i18n/zh.ts | 85 ++ 17 files changed, 3224 insertions(+), 9 deletions(-) create mode 100644 apps/desktop/src/app/chat/composer/hooks/use-status-presence.test.ts create mode 100644 apps/desktop/src/app/chat/composer/status-stack/session-control-goal.tsx create mode 100644 apps/desktop/src/app/chat/composer/status-stack/session-control-heartbeat.tsx create mode 100644 apps/desktop/src/app/chat/composer/status-stack/session-control-loop.tsx create mode 100644 apps/desktop/src/app/chat/composer/status-stack/session-control-utils.ts create mode 100644 apps/desktop/src/app/chat/composer/status-stack/session-control.test.tsx create mode 100644 apps/desktop/src/app/chat/composer/status-stack/session-control.tsx diff --git a/apps/desktop/src/app/chat/composer/hooks/use-status-presence.test.ts b/apps/desktop/src/app/chat/composer/hooks/use-status-presence.test.ts new file mode 100644 index 0000000000..1c5fb5f0b7 --- /dev/null +++ b/apps/desktop/src/app/chat/composer/hooks/use-status-presence.test.ts @@ -0,0 +1,200 @@ +import { act, cleanup, renderHook } from '@testing-library/react' +import { afterEach, beforeEach, describe, expect, it } from 'vitest' + +import { $composerActionsBySession } from '@/store/composer-actions' +import { $previewStatusBySession } from '@/store/preview-status' +import { $sessionControlBySession, type SessionControlEntry } from '@/store/session-control' +import { $todosBySession } from '@/store/todos' + +import { useSessionStatusPresence } from './use-status-presence' + +const SID = 'presence-session-1' + +const mockEntry = (overrides?: Partial): SessionControlEntry => ({ + capability: 'supported', + error: null, + loading: false, + pendingAction: null, + snapshot: { + goal: null, + heartbeat: null, + loop: null, + revision: 'rev-1', + updated_at: 1000 + }, + ...overrides +}) + +describe('useSessionStatusPresence', () => { + beforeEach(() => { + $todosBySession.set({}) + $composerActionsBySession.set({}) + $previewStatusBySession.set({}) + $sessionControlBySession.set({}) + }) + + afterEach(() => { + cleanup() + $todosBySession.set({}) + $composerActionsBySession.set({}) + $previewStatusBySession.set({}) + $sessionControlBySession.set({}) + }) + + it('returns false when session is null or empty', () => { + const { result } = renderHook(() => useSessionStatusPresence(null)) + expect(result.current).toBe(false) + + const { result: emptyResult } = renderHook(() => useSessionStatusPresence(SID)) + expect(emptyResult.current).toBe(false) + }) + + it('returns true when legacy status items exist', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $todosBySession.set({ + [SID]: [{ content: 'task 1', id: '1', status: 'in_progress' }] + }) + }) + + expect(result.current).toBe(true) + }) + + it('returns true when structured goal exists in session control', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + snapshot: { + goal: { + contract: { + boundaries: '', + constraints: '', + outcome: 'test outcome', + stop_when: '', + verification: '' + }, + gates: [], + max_turns: 10, + status: 'active', + subgoals: [], + title: 'Structured Goal', + turns_used: 1 + }, + heartbeat: null, + loop: null, + revision: 'rev-2', + updated_at: 2000 + } + }) + }) + }) + + expect(result.current).toBe(true) + }) + + it('returns true when structured loop exists in session control', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + snapshot: { + goal: null, + heartbeat: null, + loop: { + awaiting_response: false, + created_at: 1000, + current_delay: 60, + deferred_by_goal: false, + interval_seconds: 60, + last_fired_at: 1000, + max_ticks: 10, + mode: 'interval', + next_due_at: 2000, + prompt: 'Run loop', + status: 'active', + ticks_fired: 0, + times: 5, + until: '' + }, + revision: 'rev-3', + updated_at: 2000 + } + }) + }) + }) + + expect(result.current).toBe(true) + }) + + it('returns true when structured heartbeat exists in session control', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + snapshot: { + goal: null, + heartbeat: { + created_at: 1000, + fire_count: 1, + interval_seconds: 300, + last_fired_at: 1000, + prompt: 'Heartbeat check', + status: 'active' + }, + loop: null, + revision: 'rev-4', + updated_at: 2000 + } + }) + }) + }) + + expect(result.current).toBe(true) + }) + + it('returns false when session control entry has empty snapshot (null goal, loop, heartbeat)', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + snapshot: { + goal: null, + heartbeat: null, + loop: null, + revision: 'rev-empty', + updated_at: 2000 + } + }) + }) + }) + + expect(result.current).toBe(false) + }) + + it('returns true when session control entry has only an error', () => { + const { result } = renderHook(() => useSessionStatusPresence(SID)) + expect(result.current).toBe(false) + + act(() => { + $sessionControlBySession.set({ + [SID]: mockEntry({ + error: 'Gateway connection failed', + snapshot: null + }) + }) + }) + + expect(result.current).toBe(true) + }) +}) diff --git a/apps/desktop/src/app/chat/composer/hooks/use-status-presence.ts b/apps/desktop/src/app/chat/composer/hooks/use-status-presence.ts index b4655ffd50..b2c059ba35 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-status-presence.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-status-presence.ts @@ -3,8 +3,9 @@ import { useSyncExternalStore } from 'react' import { $composerActionsBySession } from '@/store/composer-actions' import { $statusItemsBySession } from '@/store/composer-status' import { $previewStatusBySession } from '@/store/preview-status' +import { $sessionControlBySession } from '@/store/session-control' -/** Structural view of the three per-session feeds — they hold different item +/** Structural view of the per-session feeds — they hold different item * types, and all this hook needs from each is "does this key have rows". */ interface PresenceFeed { get(): Record @@ -14,7 +15,7 @@ interface PresenceFeed { const FEEDS: PresenceFeed[] = [$statusItemsBySession, $composerActionsBySession, $previewStatusBySession] const subscribe = (onChange: () => void) => { - const offs = FEEDS.map(feed => feed.listen(onChange)) + const offs = [...FEEDS.map(feed => feed.listen(onChange)), $sessionControlBySession.listen(onChange)] return () => { for (const off of offs) { @@ -24,8 +25,9 @@ const subscribe = (onChange: () => void) => { } /** - * Whether a session has any status items, micro actions, or previews, as a - * coarse *edge*: the boolean only flips when the stack appears/disappears. + * Whether a session has any status items, micro actions, previews, or + * structured session controls (goal, loop, heartbeat), as a coarse *edge*: + * the boolean only flips when the stack appears/disappears. * ChatBar uses it to toggle a styling data-attr — subscribing to the whole * `$statusItemsBySession` (a `computed` that rebuilds the entire map) / * `$previewStatusBySession` maps re-rendered the ~1.4k ChatBar on every @@ -39,6 +41,16 @@ export function useSessionStatusPresence(sessionId: string | null): boolean { return false } - return FEEDS.some(feed => (feed.get()[sessionId]?.length ?? 0) > 0) + if (FEEDS.some(feed => (feed.get()[sessionId]?.length ?? 0) > 0)) { + return true + } + + const control = $sessionControlBySession.get()[sessionId] + + return Boolean( + control?.error || + (control?.snapshot && + (control.snapshot.goal !== null || control.snapshot.loop !== null || control.snapshot.heartbeat !== null)) + ) }) } diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index b01381e41d..371bf32b07 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -1195,6 +1195,7 @@ export function ChatBar({ grows upward over the thread and the dock's own measurement covers it. Collapses to nothing when every status is empty. */} 0 ? ( group.type === 'todo' && group.items.some(item => item.todoStatus === 'in_progress' && item.state === 'running') interface ComposerStatusStackProps { + onSubmit?: (value: string, options?: SubmitTextOptions) => Promise | boolean /** The queue, built by the composer (it owns the queue's callbacks). Rendered * as the last group so it stays fused to the composer like before. */ queue: ReactNode @@ -85,7 +89,7 @@ interface ComposerStatusStackProps { * every session-scoped status — subagents, background tasks, queue — grouped by * type and separated by light dividers. Collapses to nothing when empty. */ -export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackProps) { +export function ComposerStatusStack({ onSubmit, queue, sessionId }: ComposerStatusStackProps) { const { t } = useI18n() const navigate = useNavigate() // Subscribe to THIS session's slice only. Both maps churn on other @@ -96,10 +100,22 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro // items actually changed. const items = useSessionSlice($statusItemsBySession, sessionId) const previews = useSessionSlice($previewStatusBySession, sessionId) + const controlEntry = useSessionValue($sessionControlBySession, sessionId) + const scrolledUp = useStore($threadScrolledUp) const billing = useStore($billingBlock) - const groups = useMemo(() => groupStatusItems(items), [items]) + const isStructuredSupported = controlEntry?.capability === 'supported' + + const groups = useMemo(() => { + const raw = groupStatusItems(items) + + if (isStructuredSupported) { + return raw.filter(g => g.type !== 'goal') + } + + return raw + }, [items, isStructuredSupported]) // Seed from the registry on session open; event-driven refreshes (terminal / // process tool completions) live in use-message-stream. This must NOT reset @@ -112,7 +128,7 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro useEffect(() => { if (sessionId) { void refreshBackgroundProcesses(sessionId) - void refreshSessionGoal(sessionId) + void refreshSessionControl(sessionId) } }, [sessionId]) @@ -160,6 +176,21 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro sections.push({ key: 'billing', node: }) } + const hasControlContent = Boolean( + controlEntry && + (controlEntry.error || + controlEntry.snapshot?.goal || + controlEntry.snapshot?.loop || + controlEntry.snapshot?.heartbeat) + ) + + if (sessionId && controlEntry && hasControlContent) { + sections.push({ + key: 'session-control', + node: + }) + } + for (const group of groups) { sections.push({ key: group.type, diff --git a/apps/desktop/src/app/chat/composer/status-stack/session-control-goal.tsx b/apps/desktop/src/app/chat/composer/status-stack/session-control-goal.tsx new file mode 100644 index 0000000000..29781ed95c --- /dev/null +++ b/apps/desktop/src/app/chat/composer/status-stack/session-control-goal.tsx @@ -0,0 +1,606 @@ +import { memo, useCallback, useState } from 'react' + +import type { SubmitTextOptions } from '@/app/session/hooks/use-prompt-actions/utils' +import { StatusSection } from '@/components/chat/status-section' +import { Button } from '@/components/ui/button' +import { Codicon } from '@/components/ui/codicon' +import { ConfirmDialog } from '@/components/ui/confirm-dialog' +import { + ContextMenu, + ContextMenuContent, + ContextMenuItem, + ContextMenuSeparator, + ContextMenuTrigger +} from '@/components/ui/context-menu' +import { + Dialog, + DialogContent, + DialogDescription, + DialogFooter, + DialogHeader, + DialogTitle +} from '@/components/ui/dialog' +import { + DropdownMenu, + DropdownMenuContent, + DropdownMenuItem, + DropdownMenuSeparator, + DropdownMenuTrigger +} from '@/components/ui/dropdown-menu' +import { Tip } from '@/components/ui/tooltip' +import { useI18n } from '@/i18n' +import { + runSessionControlAction, + type SessionControlAction, + type SessionControlActionArgs, + type SessionControlGoal +} from '@/store/session-control' + +import type { ConfirmState } from './session-control-utils' + +interface GoalSectionProps { + goal: SessionControlGoal + sessionId: string + pendingAction: SessionControlAction | null + onSubmit?: (value: string, options?: SubmitTextOptions) => Promise | boolean + onFeedback: (error: string | null, success: string | null) => void +} + +export const SessionControlGoalSection = memo(function SessionControlGoalSection({ + goal, + sessionId, + pendingAction, + onSubmit, + onFeedback +}: GoalSectionProps) { + const { t } = useI18n() + const s = t.statusStack + const ctrl = s.control + + const [detailsOpen, setDetailsOpen] = useState(false) + const [addCriterionOpen, setAddCriterionOpen] = useState(false) + const [addCriterionError, setAddCriterionError] = useState(null) + const [newCriterionText, setNewCriterionText] = useState('') + const [confirmState, setConfirmState] = useState(null) + const [menuOpen, setMenuOpen] = useState(false) + + const isBusy = Boolean(pendingAction) + + const handleAction = useCallback( + async ( + action: SessionControlAction, + args?: SessionControlActionArgs, + onFailure?: (message: string) => void + ): Promise => { + onFeedback(null, null) + + try { + const dispatch = await runSessionControlAction(sessionId, action, args) + + if (dispatch.type === 'send') { + if (!dispatch.message || !onSubmit) { + onFeedback(ctrl.continuationFailed, null) + onFailure?.(ctrl.continuationFailed) + + return false + } + + const submitted = await onSubmit(dispatch.message, { + displayKind: 'hidden', + sessionId + }) + + if (!submitted) { + onFeedback(ctrl.continuationFailed, null) + onFailure?.(ctrl.continuationFailed) + + return false + } + } + + onFeedback(null, ctrl.actionSucceeded) + + return true + } catch (err) { + const msg = err instanceof Error ? err.message : String(err) + const failure = ctrl.actionFailed(msg) + onFeedback(failure, null) + onFailure?.(failure) + + return false + } + }, + [sessionId, onSubmit, onFeedback, ctrl] + ) + + const copyCriterionText = useCallback( + async (text: string) => { + try { + await navigator.clipboard.writeText(text) + onFeedback(null, ctrl.copySuccess) + } catch { + onFeedback(ctrl.copyFailure, null) + } + }, + [onFeedback, ctrl] + ) + + const visibleState: 'waiting' | 'active' | 'paused' | 'done' = goal.wait_barrier + ? 'waiting' + : goal.status === 'paused' + ? 'paused' + : goal.status === 'done' + ? 'done' + : 'active' + + const stateLabel = + visibleState === 'waiting' + ? s.goalWaiting + : visibleState === 'paused' + ? s.goalPaused + : visibleState === 'done' + ? s.goalDone + : s.goalActive + + const headerLabel = + visibleState === 'done' + ? `${stateLabel} · ${ctrl.goalDoneTurns(goal.turns_used)}` + : goal.max_turns > 0 + ? `${stateLabel} · ${ctrl.goalActiveTurns(goal.turns_used, goal.max_turns)}` + : goal.turns_used > 0 + ? `${stateLabel} · ${ctrl.goalTurn(goal.turns_used)}` + : stateLabel + + const confirmClearGoal = () => { + setConfirmState({ + title: ctrl.clearGoalConfirmTitle, + description: ctrl.clearGoalConfirmBody, + destructive: true, + confirmLabel: ctrl.clearGoal, + onConfirm: async () => { + await handleAction('goal.clear') + } + }) + } + + const confirmRemoveCriterion = (index: number) => { + setConfirmState({ + title: ctrl.removeCriterionConfirmTitle(index), + description: ctrl.removeCriterionConfirmBody(index), + destructive: true, + confirmLabel: ctrl.removeCriterion(index), + onConfirm: async () => { + await handleAction('subgoal.remove', { index }) + } + }) + } + + const confirmClearCriteria = () => { + setConfirmState({ + title: ctrl.clearCriteriaConfirmTitle, + description: ctrl.clearCriteriaConfirmBody, + destructive: true, + confirmLabel: ctrl.clearCriteria, + onConfirm: async () => { + await handleAction('subgoal.clear') + } + }) + } + + const openAddCriterion = () => { + setAddCriterionError(null) + setAddCriterionOpen(true) + } + + const renderMenuItems = (isContext = false) => { + const Item = isContext ? ContextMenuItem : DropdownMenuItem + const Sep = isContext ? ContextMenuSeparator : DropdownMenuSeparator + + return ( + <> + {hasDetails && ( + setDetailsOpen(true)}> + + {ctrl.viewDetails} + + )} + {visibleState !== 'done' && ( + + + {ctrl.addCriterion} + + )} + {(hasDetails || visibleState !== 'done') && } + {visibleState === 'active' && ( + void handleAction('goal.pause')}> + + {ctrl.pauseGoal} + + )} + {visibleState === 'paused' && ( + void handleAction('goal.resume')}> + + {ctrl.resumeGoal} + + )} + {visibleState === 'waiting' && ( + <> + void handleAction('goal.unwait')}> + + {ctrl.resumeNow} + + void handleAction('goal.pause')}> + + {ctrl.pauseGoal} + + + )} + + + {ctrl.clearGoal} + + + ) + } + + const hasDetails = Boolean( + goal.contract.outcome || + goal.contract.verification || + goal.contract.constraints || + goal.contract.boundaries || + goal.contract.stop_when || + goal.wait_barrier || + goal.gates.length > 0 + ) + + return ( + <> + + +
+ + + + + + + + + + {renderMenuItems(false)} + + + } + defaultCollapsed={false} + icon={} + label={headerLabel} + > +
+ {/* Full goal title */} +
{goal.title}
+ + {/* Optional reasons */} + {!detailsOpen && goal.wait_barrier && ( +
+ {goal.wait_barrier.reason + ? `${ctrl.waitBarrierTitle}: ${goal.wait_barrier.reason}` + : ctrl.waitBarrierTitle} +
+ )} + {!goal.wait_barrier && goal.paused_reason && ( +
{goal.paused_reason}
+ )} + {!goal.wait_barrier && !goal.paused_reason && goal.last_reason && ( +
{goal.last_reason}
+ )} + + {/* View details button */} + {hasDetails && ( +
+ +
+ )} + + {/* Criteria subsection */} +
+
+ {ctrl.criteriaHeader(goal.subgoals.length)} +
+ + {goal.subgoals.length > 0 && ( + + )} +
+
+ + {goal.subgoals.length > 0 ? ( +
+ {goal.subgoals.map((subgoal, idx) => { + const index = idx + 1 + + return ( +
+
+ {index}. + {subgoal} +
+
+ + + + + + +
+
+ ) + })} +
+ ) : null} +
+
+
+
+
+ {renderMenuItems(true)} +
+ + {/* Add Criterion Dialog */} + { + if (!open) { + setAddCriterionOpen(false) + setAddCriterionError(null) + } + }} + open={addCriterionOpen} + > + +
{ + e.preventDefault() + const text = newCriterionText.trim() + + if (!text || isBusy) { + return + } + + setAddCriterionError(null) + const ok = await handleAction('subgoal.add', { text }, setAddCriterionError) + + if (ok) { + setNewCriterionText('') + setAddCriterionOpen(false) + } + }} + > + + {ctrl.addCriterionDialogTitle} + {ctrl.addCriterionPlaceholder} + +
+