From d2975f4219bcad1081ced887623586caf4a4a259 Mon Sep 17 00:00:00 2001 From: Victor Kyriazakos Date: Wed, 19 Aug 2026 00:21:18 +0000 Subject: [PATCH 01/42] fix(relay): _read_loop fails in-flight pending futures on any exit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The reader is the only thing that can resolve a pending outbound_result future. When the socket dropped unexpectedly (vs a deliberate disconnect(), which already failed pending), every in-flight _request_response waiter blocked the full _outbound_timeout_s (~30s) on a future no reader could resolve. Coatue incident 2026-08-18: a 1011 keepalive ping timeout close left final-send and error-notification calls wedged against the dead socket, holding sessions active while the connector replayed inbound backlog. Fail every not-done pending future from _read_loop's finally with the dict shape callers expect ({success: False, error: connection lost}) — never an exception on the outbound path — then clear the map. list() snapshot avoids mutation-during-iteration from woken waiters' finally pops. --- gateway/relay/ws_transport.py | 97 +++++++++++-------- .../relay/test_ws_transport_hardening.py | 83 ++++++++++++++++ 2 files changed, 140 insertions(+), 40 deletions(-) create mode 100644 tests/gateway/relay/test_ws_transport_hardening.py diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index b0e067fc16..2eebdd3211 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -809,47 +809,64 @@ class WebSocketRelayTransport: assert self._ws is not None buf = "" try: - async for chunk in self._ws: - buf += chunk if isinstance(chunk, str) else chunk.decode("utf-8") - # Newline-delimited frames; keep any trailing partial line. - *lines, buf = buf.split("\n") - for line in lines: - if line.strip(): - await self._handle_frame(line) - except asyncio.CancelledError: - raise - except Exception as exc: # noqa: BLE001 - log + let the task end; reconnection handled below - # Phase 7 Unit 7d-B: detect a 4401 (unauthorized) close. After a prior - # successful handshake this is a REVOCATION (opt-out / deprovision) — - # the per-gateway secret is gone, so reconnecting is futile. Latch a - # terminal "auth revoked" state and DON'T re-dial. Before any - # successful handshake a 4401 stays retryable (cold-start race). - if self._close_code_of(exc) == _RELAY_UNAUTHORIZED_CLOSE_CODE and self._handshake_succeeded: - self._auth_revoked = True - if not self._closing: - logger.warning( - "relay ws closed 4401 (unauthorized) after a successful handshake — " - "treating as a revoked relay credential (opt-out); not reconnecting" + try: + async for chunk in self._ws: + buf += chunk if isinstance(chunk, str) else chunk.decode("utf-8") + # Newline-delimited frames; keep any trailing partial line. + *lines, buf = buf.split("\n") + for line in lines: + if line.strip(): + await self._handle_frame(line) + except asyncio.CancelledError: + raise + except Exception as exc: # noqa: BLE001 - log + let the task end; reconnection handled below + # Phase 7 Unit 7d-B: detect a 4401 (unauthorized) close. After a prior + # successful handshake this is a REVOCATION (opt-out / deprovision) — + # the per-gateway secret is gone, so reconnecting is futile. Latch a + # terminal "auth revoked" state and DON'T re-dial. Before any + # successful handshake a 4401 stays retryable (cold-start race). + if self._close_code_of(exc) == _RELAY_UNAUTHORIZED_CLOSE_CODE and self._handshake_succeeded: + self._auth_revoked = True + if not self._closing: + logger.warning( + "relay ws closed 4401 (unauthorized) after a successful handshake — " + "treating as a revoked relay credential (opt-out); not reconnecting" + ) + elif not self._closing: + logger.warning("relay ws read loop ended: %s", exc) + # Phase 5 §5.3: the socket closed. If reconnect is enabled and this was + # NOT a deliberate disconnect(), kick the reconnect supervisor so the + # gateway re-dials + re-handshakes (which triggers the connector's + # buffered-flip drain on the new handshake). Self-scheduling: the reader + # ends here, the supervisor re-dials and starts a fresh reader. + # Phase 7 Unit 7d-B: a revoked credential (terminal 4401) is the one case + # we deliberately do NOT reconnect — the secret is dead until the + # instance is recreated, so spinning would just reproduce the failure. + if ( + self._reconnect + and not self._closing + and not self._auth_revoked + and (self._supervisor is None or self._supervisor.done()) + ): + self._supervisor = asyncio.create_task( + self._reconnect_loop(), name="relay-ws-reconnect" + ) + finally: + # The reader is the ONLY thing that can resolve a pending + # outbound_result future — once it exits (socket dropped, error, + # or cancellation cleanup) every in-flight _request_response waiter + # is unresolvable and would otherwise block the full + # _outbound_timeout_s (~30s) on a dead socket (Coatue incident + # 2026-08-18: stuck sends after a 1011 keepalive close). Fail them + # NOW with the dict shape callers expect (never an exception on + # the outbound path). list() snapshot: set_result wakes waiters + # whose finally-pop would otherwise mutate the dict mid-iteration. + for _rid, fut in list(self._pending.items()): + if not fut.done(): + fut.set_result( + {"success": False, "error": "relay transport connection lost"} ) - elif not self._closing: - logger.warning("relay ws read loop ended: %s", exc) - # Phase 5 §5.3: the socket closed. If reconnect is enabled and this was - # NOT a deliberate disconnect(), kick the reconnect supervisor so the - # gateway re-dials + re-handshakes (which triggers the connector's - # buffered-flip drain on the new handshake). Self-scheduling: the reader - # ends here, the supervisor re-dials and starts a fresh reader. - # Phase 7 Unit 7d-B: a revoked credential (terminal 4401) is the one case - # we deliberately do NOT reconnect — the secret is dead until the - # instance is recreated, so spinning would just reproduce the failure. - if ( - self._reconnect - and not self._closing - and not self._auth_revoked - and (self._supervisor is None or self._supervisor.done()) - ): - self._supervisor = asyncio.create_task( - self._reconnect_loop(), name="relay-ws-reconnect" - ) + self._pending.clear() @staticmethod def _close_code_of(exc: BaseException) -> Optional[int]: diff --git a/tests/gateway/relay/test_ws_transport_hardening.py b/tests/gateway/relay/test_ws_transport_hardening.py new file mode 100644 index 0000000000..edeb141486 --- /dev/null +++ b/tests/gateway/relay/test_ws_transport_hardening.py @@ -0,0 +1,83 @@ +"""Regression tests for the relay WS transport hardening fix. + +Coatue incident 2026-08-18: WAN latency / event-loop stalls tripped the +websockets library's default 20s pong deadline, closing customer-gateway +sockets with `1011 keepalive ping timeout`. On top of the spurious close, +every in-flight outbound then hung for the full _outbound_timeout_s (~30s) +because only disconnect() failed pending futures — an unexpected socket drop +left them stranded — and sends issued while the reconnect supervisor was +backing off registered futures no reader could ever resolve. + +Three hardening changes under test: + 1. _read_loop fails all in-flight _pending futures on ANY exit path with + the dict shape callers expect ({"success": False, ...}). + 2. _request_response fails fast while the reconnect supervisor is + mid-redial (live supervisor task = the redial window). + 3. connect() passes explicit WAN-friendly keepalive tuning + (ping_interval=30, ping_timeout=60) to websockets.connect(). +""" + +from __future__ import annotations + +import asyncio + +import pytest + +import gateway.relay.ws_transport as ws_transport_mod +from gateway.relay.ws_transport import WebSocketRelayTransport, WEBSOCKETS_AVAILABLE + +pytestmark = pytest.mark.skipif(not WEBSOCKETS_AVAILABLE, reason="websockets not installed") + +if WEBSOCKETS_AVAILABLE: + from websockets.exceptions import ConnectionClosedError + + +class _DroppingWS: + """Fake socket: accepts sends, then the read loop dies mid-iteration — + the shape of an unexpected close (e.g. 1011 keepalive ping timeout).""" + + def __init__(self): + self.sent: list[str] = [] + # Reader blocks here until the test releases it, so the outbound + # future is registered BEFORE the "socket" drops. + self.drop = asyncio.Event() + + async def send(self, data): + self.sent.append(data) + + def __aiter__(self): + return self + + async def __anext__(self): + await self.drop.wait() + raise ConnectionClosedError(None, None) + + async def close(self): + pass + + +@pytest.mark.asyncio +async def test_read_loop_exit_fails_pending_futures_promptly(): + """When the socket drops unexpectedly, in-flight _request_response callers + must get {"success": False, ...} promptly — not block ~30s on a future + only the (now dead) reader could have resolved.""" + t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=30.0) + fake = _DroppingWS() + t._ws = fake + t._reader = asyncio.create_task(t._read_loop()) + + send_task = asyncio.create_task(t.send_outbound({"op": "send_message", "text": "hi"})) + # Let the outbound frame go out and its future register in _pending. + for _ in range(50): + if t._pending: + break + await asyncio.sleep(0.01) + assert t._pending, "outbound future never registered" + + # Drop the socket: the read loop exits on ConnectionClosedError. + fake.drop.set() + + result = await asyncio.wait_for(send_task, timeout=2.0) + assert result == {"success": False, "error": "relay transport connection lost"} + assert t._pending == {} + await t._reader From a7946f6502e302a12faaf38168ccf88116689cd9 Mon Sep 17 00:00:00 2001 From: Victor Kyriazakos Date: Wed, 19 Aug 2026 00:21:35 +0000 Subject: [PATCH 02/42] fix(relay): fail sends fast while the reconnect supervisor is mid-redial Between an unexpected close and the supervisor's successful re-dial, self._ws still points at the DEAD socket, so the existing None-check does not cover the backoff window: a send registered a future no reader could resolve and blocked the caller for _outbound_timeout_s. A live supervisor task is exactly the redial window (the loop returns when a dial succeeds and its fresh reader takes over), so no new state is needed: when self._supervisor is not done, return the error dict immediately ({success: False, error: reconnecting}). Callers already handle failed sends; a fast, honest failure beats a 30s wedge (Coatue 2026-08-18). --- gateway/relay/ws_transport.py | 10 ++++++++ .../relay/test_ws_transport_hardening.py | 23 +++++++++++++++++++ 2 files changed, 33 insertions(+) diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index 2eebdd3211..be7ddb71ad 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -775,6 +775,16 @@ class WebSocketRelayTransport: return {"success": False, "error": "relay transport closed"} if self._ws is None: return {"success": False, "error": "relay transport not connected"} + if self._supervisor is not None and not self._supervisor.done(): + # The reconnect supervisor is mid-redial (backing off after an + # unexpected close). `self._ws` still points at the DEAD socket + # until _dial_and_start() replaces it, so the check above does not + # cover this window — a send here would register a future no + # reader can resolve and block the caller for _outbound_timeout_s + # (Coatue incident 2026-08-18). Fail fast with the dict shape + # callers expect; a live supervisor's dial success ends the task, + # so `not done()` is exactly the redial window. + return {"success": False, "error": "relay transport reconnecting"} request_id = uuid.uuid4().hex loop = asyncio.get_running_loop() fut: asyncio.Future[Dict[str, Any]] = loop.create_future() diff --git a/tests/gateway/relay/test_ws_transport_hardening.py b/tests/gateway/relay/test_ws_transport_hardening.py index edeb141486..07991041c5 100644 --- a/tests/gateway/relay/test_ws_transport_hardening.py +++ b/tests/gateway/relay/test_ws_transport_hardening.py @@ -81,3 +81,26 @@ async def test_read_loop_exit_fails_pending_futures_promptly(): assert result == {"success": False, "error": "relay transport connection lost"} assert t._pending == {} await t._reader + + +@pytest.mark.asyncio +async def test_send_during_redial_window_fails_fast(): + """While the reconnect supervisor is backing off, _ws still points at the + dead socket — a send must return an error dict immediately (no + RuntimeError, no 30s timeout on an unresolvable future).""" + t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=30.0) + t._ws = _DroppingWS() # stale/dead socket left over from before the drop + t._supervisor = asyncio.create_task(asyncio.sleep(60)) # mid-redial backoff + try: + result = await asyncio.wait_for( + t.send_outbound({"op": "send_message", "text": "hi"}), timeout=1.0 + ) + assert result["success"] is False + assert "reconnect" in result["error"] + assert t._pending == {} + finally: + t._supervisor.cancel() + try: + await t._supervisor + except asyncio.CancelledError: + pass From 534f8d4fb098b36b5ef5ea31fe401209faa023c4 Mon Sep 17 00:00:00 2001 From: Victor Kyriazakos Date: Wed, 19 Aug 2026 00:22:01 +0000 Subject: [PATCH 03/42] fix(relay): WAN-friendly keepalive tuning (ping_interval=30, ping_timeout=60) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Customer gateways cross WAN paths to the connector; the websockets library defaults (ping 20s, pong deadline 20s) close the socket with 1011 keepalive ping timeout under transient latency or event-loop stalls — the trigger of the Coatue 2026-08-18 incident. 60s tolerates stalls while still detecting a genuinely dead link within ~90s worst case. Set explicitly at both connect() call sites (with and without auth headers) so the tuning is visible and pinned by test rather than inherited from library defaults. --- gateway/relay/ws_transport.py | 20 +++++++- .../relay/test_ws_transport_hardening.py | 48 +++++++++++++++++++ 2 files changed, 66 insertions(+), 2 deletions(-) diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index be7ddb71ad..880a28c32f 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -497,10 +497,26 @@ class WebSocketRelayTransport: # normal fast backoff, not the dormant cadence. self._dormant = False headers = self._upgrade_headers() + # WAN-friendly keepalive: customer gateways cross WAN paths to the + # connector; the websockets library default (ping_interval=20, + # ping_timeout=20) gives the peer only a 20s pong deadline, which + # produces spurious `1011 keepalive ping timeout` closes under + # transient latency / event-loop stalls (Coatue incident 2026-08-18). + # ping_timeout=60 tolerates such stalls while still detecting a dead + # link within ~90s worst case (30s interval + 60s pong deadline). if headers: - self._ws = await websockets.connect(self._url, additional_headers=headers) # type: ignore[union-attr] + self._ws = await websockets.connect( # type: ignore[union-attr] + self._url, + additional_headers=headers, + ping_interval=30, + ping_timeout=60, + ) else: - self._ws = await websockets.connect(self._url) # type: ignore[union-attr] + self._ws = await websockets.connect( # type: ignore[union-attr] + self._url, + ping_interval=30, + ping_timeout=60, + ) self._reader = asyncio.create_task(self._read_loop(), name="relay-ws-reader") # Send one hello PER fronted identity (Phase 1.5 Shape A). The connector # accumulates them into its advertised set (the first sets the session diff --git a/tests/gateway/relay/test_ws_transport_hardening.py b/tests/gateway/relay/test_ws_transport_hardening.py index 07991041c5..b87ff8f9ea 100644 --- a/tests/gateway/relay/test_ws_transport_hardening.py +++ b/tests/gateway/relay/test_ws_transport_hardening.py @@ -104,3 +104,51 @@ async def test_send_during_redial_window_fails_fast(): await t._supervisor except asyncio.CancelledError: pass + + +@pytest.mark.asyncio +async def test_connect_passes_wan_keepalive_tuning(monkeypatch): + """connect() must pass ping_interval=30 / ping_timeout=60 explicitly — + the library defaults (20/20) caused spurious 1011 keepalive closes over + WAN paths (Coatue 2026-08-18). Both call sites (with/without auth + headers) are exercised.""" + captured: list[dict] = [] + + class _IdleWS: + async def send(self, data): + pass + + def __aiter__(self): + return self + + async def __anext__(self): + await asyncio.sleep(3600) + + async def close(self): + pass + + async def _fake_connect(url, **kwargs): + captured.append(kwargs) + return _IdleWS() + + monkeypatch.setattr(ws_transport_mod.websockets, "connect", _fake_connect) + + # Site 1: no upgrade secret -> the headerless connect() call. + t = WebSocketRelayTransport("ws://unused", "discord", "bot1") + await t.connect() + await t.disconnect(budget_s=0) + + # Site 2: secret + gateway_id -> the additional_headers connect() call. + t2 = WebSocketRelayTransport( + "ws://unused", "discord", "bot1", gateway_id="gw-1", upgrade_secret="s3cret" + ) + await t2.connect() + await t2.disconnect(budget_s=0) + + assert len(captured) == 2 + no_header_kwargs, header_kwargs = captured + assert "additional_headers" not in no_header_kwargs + assert "additional_headers" in header_kwargs + for kwargs in captured: + assert kwargs.get("ping_interval") == 30 + assert kwargs.get("ping_timeout") == 60 From b5f95d0890bebe3d54ea225d9ffd6246fd38c7b2 Mon Sep 17 00:00:00 2001 From: Victor Kyriazakos Date: Sat, 15 Aug 2026 12:30:13 +0000 Subject: [PATCH 04/42] =?UTF-8?q?fix(relay):=20drop=20replayed=20inbounds?= =?UTF-8?q?=20=E2=80=94=20bounded=20dedupe=20on=20(chat=5Fid,=20message=5F?= =?UTF-8?q?id)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Live-canary finding #3 (Alice, staging): the relay inbound leg is at-least-once. On WS re-handshake the connector replays its durable per-instance buffer; a long multi-tool turn (60-100s) straddling a quiet socket drop got its ORIGINAL inbound replayed after the turn finished, re-running the entire turn — the user saw the final answer posted 2-5x (each a separate execution, hence slightly different texts). Receipts: same msg text at history=0 in back-to-back sessions 121647/121840, no Slack-side retry on the connector (envelope dedupe never fired). Consumer-side idempotency: bounded FIFO seen-set (512) keyed by platform message identity; events without a message_id never dedupe (fail-open — dropping a real message is worse than rerunning one). No wire change; contract v1 untouched. Transplanted-from: victor-fork/feat/relay-slack-live-cards@73ce04ae75 (extracted for the rc.4 relay-fixes train; tests moved to a standalone file with no live-cards dependencies) --- gateway/relay/adapter.py | 37 ++++++ .../relay/test_relay_inbound_dedupe.py | 123 ++++++++++++++++++ 2 files changed, 160 insertions(+) create mode 100644 tests/gateway/relay/test_relay_inbound_dedupe.py diff --git a/gateway/relay/adapter.py b/gateway/relay/adapter.py index 10cec8775d..9621827781 100644 --- a/gateway/relay/adapter.py +++ b/gateway/relay/adapter.py @@ -95,6 +95,9 @@ class RelayAdapter(BasePlatformAdapter): # feedback off SendResult — see send()). Consumed by the gateway's # semantic thread-rename lane; bounded like the sibling caches. self._auto_thread_by_chat: Dict[str, Tuple[str, str]] = {} + # Bounded FIFO seen-set for inbound replay dedupe (finding #3); + # dict preserves insertion order, giving cheap oldest-first eviction. + self._seen_inbound: Dict[str, None] = {} # chat_id -> event fired when the entry above lands, so a consumer that # arrives before the send can wait for it instead of polling. See # wait_for_auto_thread_info. @@ -361,6 +364,24 @@ class RelayAdapter(BasePlatformAdapter): async def _on_inbound(self, event) -> None: """Bridge a connector-delivered MessageEvent into the normal adapter path.""" + # Inbound replay dedupe (live-canary finding #3, Alice staging): the + # relay leg is at-least-once — on WS re-handshake the connector + # replays its durable per-instance buffer, and a long multi-tool turn + # (60-100s) straddling a quiet socket drop gets its ORIGINAL inbound + # replayed after the turn completes, re-running the whole turn (user + # saw the final answer 2-5x). Platform message identity (chat_id + + # message_id/ts) is stable across replays, so a bounded seen-set + # drops them. Consumer-side idempotency; no wire change. + dedupe_key = self._inbound_dedupe_key(event) + if dedupe_key is not None: + if dedupe_key in self._seen_inbound: + logger.info( + "relay inbound dropped as replay (dedupe key=%s)", dedupe_key + ) + return + self._seen_inbound[dedupe_key] = None + while len(self._seen_inbound) > self._SEEN_INBOUND_MAX: + self._seen_inbound.pop(next(iter(self._seen_inbound))) self._capture_scope(event) self._stamp_slack_session_thread(event) # Phase 3: a structured prompt answer resolves its waiting primitive @@ -372,6 +393,22 @@ class RelayAdapter(BasePlatformAdapter): await self._localize_inbound_media(event) await self.handle_message(event) + _SEEN_INBOUND_MAX = 512 + + def _inbound_dedupe_key(self, event) -> Optional[str]: + """Stable replay identity: (chat, platform message id). + + Returns None when the event carries no platform message id (synthetic + events, some prompt responses) — those never dedupe, fail-open by + design: dropping a real user message is strictly worse than rerunning + one, so only dedupe when identity is certain. + """ + message_id = getattr(event, "message_id", None) + chat_id = getattr(event, "chat_id", None) + if not message_id or not chat_id: + return None + return f"{chat_id}:{message_id}" + def _relay_slack_extra(self) -> Dict[str, Any]: """The Slack-behavior subset of the RELAY platform config. diff --git a/tests/gateway/relay/test_relay_inbound_dedupe.py b/tests/gateway/relay/test_relay_inbound_dedupe.py new file mode 100644 index 0000000000..ac7b8c7f92 --- /dev/null +++ b/tests/gateway/relay/test_relay_inbound_dedupe.py @@ -0,0 +1,123 @@ +"""Inbound replay dedupe on the relay adapter (transplanted from the +live-cards branch for the rc.4 relay-fixes train). + +Live-canary finding #3 (Alice, staging): the relay inbound leg is +at-least-once. On WS re-handshake the connector replays its durable +per-instance buffer; a long multi-tool turn straddling a quiet socket drop +got its ORIGINAL inbound replayed after the turn finished, re-running the +entire turn — the user saw the final answer posted 2-5x. Platform message +identity (chat_id + message_id/ts) is stable across replays, so a bounded +seen-set drops them. Fail-open: events without a message_id never dedupe +(dropping a real message is strictly worse than rerunning one). +""" + +from __future__ import annotations + +import asyncio +from types import SimpleNamespace + +import pytest + +from gateway.config import PlatformConfig +from gateway.relay.adapter import RelayAdapter +from gateway.relay.descriptor import CONTRACT_VERSION, CapabilityDescriptor +from tests.gateway.relay.stub_connector import StubConnector + + +def make_desc(**kw) -> CapabilityDescriptor: + base = dict( + contract_version=CONTRACT_VERSION, + platform="slack", + label="Slack", + max_message_length=39000, + supports_draft_streaming=True, + supports_edit=True, + supports_threads=True, + markdown_dialect="slack", + len_unit="chars", + emoji="\U0001f4ac", + platform_hint="", + pii_safe=False, + supported_ops=("send", "edit", "typing"), + ) + base.update(kw) + return CapabilityDescriptor(**base) + + +def _connected_adapter(**desc_kw): + desc = make_desc(**desc_kw) + stub = StubConnector(desc) + adapter = RelayAdapter(PlatformConfig(), desc, transport=stub) + return adapter, stub + + +@pytest.fixture() +def loop(): + loop = asyncio.new_event_loop() + asyncio.set_event_loop(loop) + yield loop + loop.close() + + +def _record(bucket, event): + async def _coro(): + bucket.append(event) + return _coro() + + +async def _false_coro(): + return False + + +async def _none_coro(): + return None + + +class TestInboundReplayDedupe: + """Finding #3 (live canary): connector replay of the original inbound + after a WS re-handshake must not re-run the turn.""" + + def _event(self, message_id="1700.100", chat_id="C1", text="hi"): + return SimpleNamespace( + message_id=message_id, chat_id=chat_id, text=text, media=None + ) + + def _tap(self, adapter, handled): + adapter.handle_message = lambda e: _record(handled, e) + adapter._consume_prompt_response = lambda e: _false_coro() + adapter._localize_inbound_media = lambda e: _none_coro() + + def test_replayed_inbound_dropped(self, loop): + adapter, _ = _connected_adapter() + handled = [] + self._tap(adapter, handled) + e = self._event() + loop.run_until_complete(adapter._on_inbound(e)) + loop.run_until_complete(adapter._on_inbound(e)) # replay + assert len(handled) == 1 + + def test_distinct_messages_both_handled(self, loop): + adapter, _ = _connected_adapter() + handled = [] + self._tap(adapter, handled) + loop.run_until_complete(adapter._on_inbound(self._event("1700.100"))) + loop.run_until_complete(adapter._on_inbound(self._event("1700.200"))) + assert len(handled) == 2 + + def test_missing_message_id_fails_open(self, loop): + adapter, _ = _connected_adapter() + handled = [] + self._tap(adapter, handled) + e = self._event(message_id=None) + loop.run_until_complete(adapter._on_inbound(e)) + loop.run_until_complete(adapter._on_inbound(e)) + assert len(handled) == 2 # never dedupe without identity + + def test_seen_set_bounded(self, loop): + adapter, _ = _connected_adapter() + adapter.handle_message = lambda e: _none_coro() + adapter._consume_prompt_response = lambda e: _false_coro() + adapter._localize_inbound_media = lambda e: _none_coro() + for i in range(600): + loop.run_until_complete(adapter._on_inbound(self._event(f"ts.{i}"))) + assert len(adapter._seen_inbound) <= adapter._SEEN_INBOUND_MAX From 22dfdc1c56a419722395d875b4d09661e1918082 Mon Sep 17 00:00:00 2001 From: Ben Barclay Date: Thu, 20 Aug 2026 14:07:11 +1000 Subject: [PATCH 05/42] fix(relay): dedupe key reads chat identity from event.source, keyed per platform MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The dedupe key read event.chat_id — a field MessageEvent does not have (chat identity lives on event.source.chat_id; see how every other read in this adapter resolves it). getattr defaulted to None, so _inbound_dedupe_key returned None for EVERY production event: the fail-open branch always taken, the seen-set permanently empty, and a replayed inbound still re-ran the whole turn. The tests passed because their SimpleNamespace events carried a top-level chat_id no production code path produces. Read the chat id from event.source, and join the underlying platform into the key: one relay adapter fronts several platforms (Phase 1.5 multiplex), and two platforms' numeric chat/message ids must not collide into one replay identity. The test event factory now builds real MessageEvent/SessionSource objects in the wire decoder's shape, so the replay and bounded-set tests fail against the broken key instead of green-lighting it. --- gateway/relay/adapter.py | 14 +++++++++++--- .../gateway/relay/test_relay_inbound_dedupe.py | 18 ++++++++++++++---- 2 files changed, 25 insertions(+), 7 deletions(-) diff --git a/gateway/relay/adapter.py b/gateway/relay/adapter.py index 9621827781..2101a8faff 100644 --- a/gateway/relay/adapter.py +++ b/gateway/relay/adapter.py @@ -396,18 +396,26 @@ class RelayAdapter(BasePlatformAdapter): _SEEN_INBOUND_MAX = 512 def _inbound_dedupe_key(self, event) -> Optional[str]: - """Stable replay identity: (chat, platform message id). + """Stable replay identity: (platform, chat, platform message id). + + Chat identity lives on ``event.source`` (MessageEvent has no top-level + ``chat_id``), and this adapter can front SEVERAL platforms over one + relay socket (Phase 1.5 multiplex), so the underlying platform joins + the key — two platforms' numeric chat/message ids must never collide + into one identity. Returns None when the event carries no platform message id (synthetic events, some prompt responses) — those never dedupe, fail-open by design: dropping a real user message is strictly worse than rerunning one, so only dedupe when identity is certain. """ + source = getattr(event, "source", None) message_id = getattr(event, "message_id", None) - chat_id = getattr(event, "chat_id", None) + chat_id = getattr(source, "chat_id", None) if not message_id or not chat_id: return None - return f"{chat_id}:{message_id}" + platform = getattr(getattr(source, "platform", None), "value", "") + return f"{platform}:{chat_id}:{message_id}" def _relay_slack_extra(self) -> Dict[str, Any]: """The Slack-behavior subset of the RELAY platform config. diff --git a/tests/gateway/relay/test_relay_inbound_dedupe.py b/tests/gateway/relay/test_relay_inbound_dedupe.py index ac7b8c7f92..561537094e 100644 --- a/tests/gateway/relay/test_relay_inbound_dedupe.py +++ b/tests/gateway/relay/test_relay_inbound_dedupe.py @@ -14,11 +14,11 @@ seen-set drops them. Fail-open: events without a message_id never dedupe from __future__ import annotations import asyncio -from types import SimpleNamespace import pytest -from gateway.config import PlatformConfig +from gateway.config import Platform, PlatformConfig +from gateway.platforms.base import MessageEvent, SessionSource from gateway.relay.adapter import RelayAdapter from gateway.relay.descriptor import CONTRACT_VERSION, CapabilityDescriptor from tests.gateway.relay.stub_connector import StubConnector @@ -78,9 +78,19 @@ class TestInboundReplayDedupe: after a WS re-handshake must not re-run the turn.""" def _event(self, message_id="1700.100", chat_id="C1", text="hi"): - return SimpleNamespace( - message_id=message_id, chat_id=chat_id, text=text, media=None + # A REAL MessageEvent, shaped exactly as _event_from_wire produces it: + # chat identity lives on event.source, NOT as a top-level attribute. + # (The first version of these tests used a SimpleNamespace with a + # top-level chat_id — a shape no production code path produces — and + # green-lit a dedupe key that read the wrong field.) + source = SessionSource( + platform=Platform.SLACK, + chat_id=chat_id, + chat_type="channel", + user_id="U1", + message_id=message_id, ) + return MessageEvent(text=text, source=source, message_id=message_id) def _tap(self, adapter, handled): adapter.handle_message = lambda e: _record(handled, e) From 8772702503f7055d12db778e53e75cd65ea6577f Mon Sep 17 00:00:00 2001 From: Ben Barclay Date: Thu, 20 Aug 2026 14:08:36 +1000 Subject: [PATCH 06/42] test(relay): dedupe regression tests drive the real wire-decode path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The dedupe tests validated hand-built events only, so a key that read a field the production event type doesn't have still went green — and the 'all 7 new tests fail against base' mutation claim didn't hold either (the fail-open and distinct-message tests pass against base because a no-op dedupe trivially satisfies both). Add a wire-level class that decodes a connector frame with _event_from_wire and dispatches it through _on_inbound — the exact production path — asserting: a decoded event yields a dedupe key at all, a re-delivered frame is dropped (fresh decode each time, so identity must come from the key, not the object), and identical chat/message ids on two different platforms are NOT conflated (Phase 1.5 multiplex). Mutation-checked: with the previous event-shape key reinstated, the wire tests fail (3 failed); with the fix, all 7 pass. --- .../relay/test_relay_inbound_dedupe.py | 69 +++++++++++++++++++ 1 file changed, 69 insertions(+) diff --git a/tests/gateway/relay/test_relay_inbound_dedupe.py b/tests/gateway/relay/test_relay_inbound_dedupe.py index 561537094e..0fd717fae1 100644 --- a/tests/gateway/relay/test_relay_inbound_dedupe.py +++ b/tests/gateway/relay/test_relay_inbound_dedupe.py @@ -131,3 +131,72 @@ class TestInboundReplayDedupe: for i in range(600): loop.run_until_complete(adapter._on_inbound(self._event(f"ts.{i}"))) assert len(adapter._seen_inbound) <= adapter._SEEN_INBOUND_MAX + + +class TestWireLevelReplayDedupe: + """The full production inbound path: a connector wire frame decoded by + _event_from_wire, then dispatched through RelayAdapter._on_inbound. + + This is the layer the hand-built-event tests above cannot vouch for: the + dedupe key must work on the exact object shape the wire decoder emits. + The original dedupe commit shipped green on hand-built events while being + a no-op on decoded ones — this class exists so that can't recur. + """ + + WIRE = { + "text": "hi", + "message_type": "text", + "message_id": "1700.100", + "source": { + "platform": "slack", + "chat_id": "C1", + "chat_type": "channel", + "user_id": "U1", + "message_id": "1700.100", + }, + } + + def _tap(self, adapter, handled): + adapter.handle_message = lambda e: _record(handled, e) + adapter._consume_prompt_response = lambda e: _false_coro() + adapter._localize_inbound_media = lambda e: _none_coro() + + def _decode(self, **overrides): + from gateway.relay.ws_transport import _event_from_wire + + raw = {**self.WIRE, **overrides} + if "source" in overrides: + raw["source"] = {**self.WIRE["source"], **overrides["source"]} + return _event_from_wire(raw) + + def test_decoded_event_yields_a_dedupe_key(self): + adapter, _ = _connected_adapter() + key = adapter._inbound_dedupe_key(self._decode()) + assert key is not None, ( + "the wire decoder's event shape must produce a dedupe key — " + "None here means the dedupe is fail-open for ALL production " + "traffic (the original ship-broken state)" + ) + + def test_replayed_wire_frame_dropped(self, loop): + adapter, _ = _connected_adapter() + handled = [] + self._tap(adapter, handled) + loop.run_until_complete(adapter._on_inbound(self._decode())) + # The connector re-delivers the SAME frame on re-handshake; the + # decoder builds a fresh object each time, so identity must come + # from the key, not object identity. + loop.run_until_complete(adapter._on_inbound(self._decode())) + assert len(handled) == 1 + + def test_same_ids_on_different_platforms_not_conflated(self, loop): + # Phase 1.5 multiplex: one adapter fronts several platforms. Numeric + # chat/message ids can collide across platforms; both must dispatch. + adapter, _ = _connected_adapter() + handled = [] + self._tap(adapter, handled) + loop.run_until_complete(adapter._on_inbound(self._decode())) + loop.run_until_complete( + adapter._on_inbound(self._decode(source={"platform": "discord"})) + ) + assert len(handled) == 2 From 52f9979a6c2dff44048cde6f99cd187f04872e9b Mon Sep 17 00:00:00 2001 From: Ben Barclay Date: Thu, 20 Aug 2026 14:10:54 +1000 Subject: [PATCH 07/42] fix(relay): reader clears the dead socket handle on unexpected exit MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two reader-exit paths arm NO reconnect supervisor: a terminal 4401 revocation (deliberately never re-dials) and reconnect=False transports. On both, _ws kept pointing at the dead socket after the reader unwound, so the 'is None' liveness guard reported connected and a send registered a future nothing could ever resolve — the full _outbound_timeout_s (~30s) wedge this PR exists to eliminate. The revocation case is the sharpest: the fatal-error notification that path emits is itself an outbound send, so it ate the stall. Null the handle in the reader's finally, identity-guarded (only if _ws still points at the socket THIS reader served) so a supervisor re-dial that already installed a fresh socket is never clobbered, and gated on not _closing so disconnect() keeps sole ownership of teardown. This also makes _ws the single honest liveness signal for the redial window itself — groundwork for retiring the supervisor-state send guard, which misreads that window from both directions. Regression tests cover both uncovered paths, asserting _ws is cleared and a post-drop send fails fast; both wedge (fail) with the finally reverted. --- gateway/relay/ws_transport.py | 17 +++++ .../relay/test_ws_transport_hardening.py | 63 ++++++++++++++++++- 2 files changed, 79 insertions(+), 1 deletion(-) diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index 880a28c32f..9b82867237 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -833,6 +833,10 @@ class WebSocketRelayTransport: async def _read_loop(self) -> None: assert self._ws is not None + # Bind the socket this reader serves: the finally below must only + # clear _ws if it still points at THIS socket (a supervisor re-dial + # may have already installed a fresh one by the time we unwind). + ws = self._ws buf = "" try: try: @@ -878,6 +882,19 @@ class WebSocketRelayTransport: self._reconnect_loop(), name="relay-ws-reconnect" ) finally: + # The socket this reader served is dead. Drop the handle (identity- + # guarded: a re-dial that already installed a FRESH socket must not + # be clobbered) so every `self._ws is None` liveness check — send, + # _request_response, go_idle, go_dormant — reports "not connected" + # for the whole outage. Without this, _ws kept pointing at the dead + # socket on every reader exit that arms NO supervisor (terminal + # 4401 revocation, reconnect=False transports), and a send there + # registered a future nothing could resolve: a full + # _outbound_timeout_s (~30s) wedge — including the revocation + # path's own fatal-error notification. disconnect() owns the + # handle during deliberate teardown, so leave it alone then. + if self._ws is ws and not self._closing: + self._ws = None # The reader is the ONLY thing that can resolve a pending # outbound_result future — once it exits (socket dropped, error, # or cancellation cleanup) every in-flight _request_response waiter diff --git a/tests/gateway/relay/test_ws_transport_hardening.py b/tests/gateway/relay/test_ws_transport_hardening.py index b87ff8f9ea..ccb3642305 100644 --- a/tests/gateway/relay/test_ws_transport_hardening.py +++ b/tests/gateway/relay/test_ws_transport_hardening.py @@ -36,11 +36,12 @@ class _DroppingWS: """Fake socket: accepts sends, then the read loop dies mid-iteration — the shape of an unexpected close (e.g. 1011 keepalive ping timeout).""" - def __init__(self): + def __init__(self, close_code: int | None = None): self.sent: list[str] = [] # Reader blocks here until the test releases it, so the outbound # future is registered BEFORE the "socket" drops. self.drop = asyncio.Event() + self._close_code = close_code async def send(self, data): self.sent.append(data) @@ -50,6 +51,10 @@ class _DroppingWS: async def __anext__(self): await self.drop.wait() + if self._close_code is not None: + from websockets.frames import Close + + raise ConnectionClosedError(Close(self._close_code, ""), None) raise ConnectionClosedError(None, None) async def close(self): @@ -152,3 +157,59 @@ async def test_connect_passes_wan_keepalive_tuning(monkeypatch): for kwargs in captured: assert kwargs.get("ping_interval") == 30 assert kwargs.get("ping_timeout") == 60 + + +async def _run_reader_to_exit(t: WebSocketRelayTransport, fake: _DroppingWS) -> None: + """Start the reader on ``fake``, drop the socket, and wait for the reader + to fully unwind — the state every post-drop assertion depends on.""" + t._reader = asyncio.create_task(t._read_loop()) + await asyncio.sleep(0) + fake.drop.set() + await t._reader + + +@pytest.mark.asyncio +async def test_send_after_terminal_4401_revocation_fails_fast(): + """A terminal 4401 revocation deliberately arms NO reconnect supervisor, + so the reader's exit is the LAST liveness transition this transport will + ever make. If _ws still points at the dead socket afterwards, the + revocation path's own fatal-error notification send wedges for the full + _outbound_timeout_s. The reader must leave _ws cleared so the + not-connected guard answers instantly.""" + t = WebSocketRelayTransport( + "ws://unused", "discord", "bot1", reconnect=True, outbound_timeout_s=30.0 + ) + fake = _DroppingWS(close_code=4401) + t._ws = fake + t._handshake_succeeded = True # prior handshake -> 4401 is a revocation + await _run_reader_to_exit(t, fake) + + assert t._auth_revoked is True + assert t._supervisor is None # revocation must not re-dial + assert t._ws is None, "dead socket handle must not survive the reader" + + result = await asyncio.wait_for( + t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0 + ) + assert result["success"] is False + assert t._pending == {} + + +@pytest.mark.asyncio +async def test_send_after_drop_with_reconnect_disabled_fails_fast(): + """reconnect=False transports never arm a supervisor either — the same + stranded-_ws wedge as the revocation path, reachable by configuration.""" + t = WebSocketRelayTransport( + "ws://unused", "discord", "bot1", reconnect=False, outbound_timeout_s=30.0 + ) + fake = _DroppingWS() + t._ws = fake + await _run_reader_to_exit(t, fake) + + assert t._ws is None, "dead socket handle must not survive the reader" + + result = await asyncio.wait_for( + t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0 + ) + assert result["success"] is False + assert t._pending == {} From ca3438d395e7536840d8fe7633b38075d85cbfc5 Mon Sep 17 00:00:00 2001 From: Ben Barclay Date: Thu, 20 Aug 2026 14:12:33 +1000 Subject: [PATCH 08/42] fix(relay): gate sends on the socket handle, not supervisor state MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The mid-redial fail-fast guard used 'supervisor task not done' as the definition of the redial window, and that signal is wrong from both directions: - Too narrow: the wedge it fixed also occurs on reader exits that arm NO supervisor (terminal 4401 revocation, reconnect=False) — those stayed wedged. - Too broad: _reconnect_loop -> _dial_and_start installs the fresh socket and starts its reader, THEN awaits one hello send per fronted identity before the supervisor unwinds. Through those awaits the transport is fully live, yet the guard rejected every send as 'reconnecting' — refusing real traffic on a healthy socket. With the previous commit the reader clears _ws on unexpected exit, so the existing 'is None' check now covers the entire outage window honestly: _ws is the single liveness signal. Drop the supervisor-state guard. The redial-window test now drives the real sequence (reader exit arms the supervisor and clears _ws) instead of hand-crafting a stale-_ws state the transport can no longer reach, and a new test pins the post-dial window: a send issued while the supervisor is still unwinding past a live socket must reach that socket (fails with the guard reinstated). --- gateway/relay/ws_transport.py | 10 -- .../relay/test_ws_transport_hardening.py | 117 ++++++++++++++++-- 2 files changed, 106 insertions(+), 21 deletions(-) diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index 9b82867237..e4924992c5 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -791,16 +791,6 @@ class WebSocketRelayTransport: return {"success": False, "error": "relay transport closed"} if self._ws is None: return {"success": False, "error": "relay transport not connected"} - if self._supervisor is not None and not self._supervisor.done(): - # The reconnect supervisor is mid-redial (backing off after an - # unexpected close). `self._ws` still points at the DEAD socket - # until _dial_and_start() replaces it, so the check above does not - # cover this window — a send here would register a future no - # reader can resolve and block the caller for _outbound_timeout_s - # (Coatue incident 2026-08-18). Fail fast with the dict shape - # callers expect; a live supervisor's dial success ends the task, - # so `not done()` is exactly the redial window. - return {"success": False, "error": "relay transport reconnecting"} request_id = uuid.uuid4().hex loop = asyncio.get_running_loop() fut: asyncio.Future[Dict[str, Any]] = loop.create_future() diff --git a/tests/gateway/relay/test_ws_transport_hardening.py b/tests/gateway/relay/test_ws_transport_hardening.py index ccb3642305..88e5b151e1 100644 --- a/tests/gateway/relay/test_ws_transport_hardening.py +++ b/tests/gateway/relay/test_ws_transport_hardening.py @@ -90,26 +90,121 @@ async def test_read_loop_exit_fails_pending_futures_promptly(): @pytest.mark.asyncio async def test_send_during_redial_window_fails_fast(): - """While the reconnect supervisor is backing off, _ws still points at the - dead socket — a send must return an error dict immediately (no - RuntimeError, no 30s timeout on an unresolvable future).""" - t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=30.0) - t._ws = _DroppingWS() # stale/dead socket left over from before the drop - t._supervisor = asyncio.create_task(asyncio.sleep(60)) # mid-redial backoff + """While the reconnect supervisor is backing off after a drop, a send must + return an error dict immediately (no RuntimeError, no 30s timeout on an + unresolvable future). Drives the REAL sequence — reader exit arms the + supervisor and clears _ws — rather than hand-crafting a stale-_ws state + the transport can no longer reach.""" + t = WebSocketRelayTransport( + "ws://unused", + "discord", + "bot1", + reconnect=True, + reconnect_backoff_s=60.0, # park the supervisor in backoff + outbound_timeout_s=30.0, + ) + fake = _DroppingWS() + t._ws = fake + await _run_reader_to_exit(t, fake) + supervisor = t._supervisor try: + assert supervisor is not None and not supervisor.done(), ( + "reader exit must arm the reconnect supervisor" + ) result = await asyncio.wait_for( t.send_outbound({"op": "send_message", "text": "hi"}), timeout=1.0 ) assert result["success"] is False - assert "reconnect" in result["error"] assert t._pending == {} finally: - t._supervisor.cancel() - try: - await t._supervisor - except asyncio.CancelledError: + if supervisor is not None: + supervisor.cancel() + try: + await supervisor + except asyncio.CancelledError: + pass + + +@pytest.mark.asyncio +async def test_send_allowed_once_redial_installs_fresh_socket(monkeypatch): + """The moment _dial_and_start() installs a fresh socket and its reader, + the transport is genuinely usable — even though the supervisor task has + not finished unwinding (it is still awaiting the hello sends). A send in + that window must be ACCEPTED, not rejected as 'reconnecting': gating + sends on supervisor state rejected real traffic on a live socket.""" + + class _LiveWS: + def __init__(self): + self.sent: list[str] = [] + self.hello_seen = asyncio.Event() + self.release = asyncio.Event() + + async def send(self, data): + self.sent.append(data) + if '"hello"' in data: + # Inside _dial_and_start, AFTER _ws and the reader are + # installed. Park here to hold the window open. + self.hello_seen.set() + await self.release.wait() + + def __aiter__(self): + return self + + async def __anext__(self): + await asyncio.sleep(3600) + + async def close(self): pass + live = _LiveWS() + + async def _fake_connect(url, **kwargs): + return live + + monkeypatch.setattr(ws_transport_mod.websockets, "connect", _fake_connect) + + t = WebSocketRelayTransport( + "ws://unused", + "discord", + "bot1", + reconnect=True, + reconnect_backoff_s=0.01, + outbound_timeout_s=5.0, + ) + # Arm the supervisor exactly as the reader's fall-through does. + t._supervisor = asyncio.create_task(t._reconnect_loop()) + await asyncio.wait_for(live.hello_seen.wait(), timeout=2.0) + try: + assert t._ws is live and not t._supervisor.done() + + send_task = asyncio.create_task( + t.send_outbound({"op": "send_message", "text": "hi"}) + ) + # The send must reach the live socket (registered + frame written), + # not fail fast: wait for the outbound frame to land. + for _ in range(100): + if any('"outbound"' in s for s in live.sent): + break + await asyncio.sleep(0.01) + assert any('"outbound"' in s for s in live.sent), ( + "send was rejected during the post-dial window despite a live " + "socket and running reader" + ) + + # Resolve it via the reader path shape: answer directly. + rid = next(iter(t._pending)) + t._pending[rid].set_result({"success": True}) + assert (await asyncio.wait_for(send_task, timeout=2.0)) == {"success": True} + finally: + live.release.set() + await asyncio.wait_for(t._supervisor, timeout=2.0) + if t._reader is not None: + t._reader.cancel() + try: + await t._reader + except asyncio.CancelledError: + pass + @pytest.mark.asyncio async def test_connect_passes_wan_keepalive_tuning(monkeypatch): From 2f77ca19932a91b37164da9dfd49eaaa4892a817 Mon Sep 17 00:00:00 2001 From: Ben Barclay Date: Thu, 20 Aug 2026 14:13:18 +1000 Subject: [PATCH 09/42] fix(relay): reader without a socket settles pending waiters instead of asserting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _read_loop opened with 'assert self._ws is not None' — an exit that escaped BEFORE the finally that fails pending futures, contradicting the 'fails pending on ANY exit path' invariant the hardening commit established. Production currently assigns _ws before scheduling the reader, so this was latent, but any future lifecycle change hitting it would strand every in-flight waiter for the full outbound timeout with only an AssertionError in the logs. Turn it into a guarded early-return INSIDE the try: the reader logs the lifecycle bug and unwinds through the same finally as every other exit, settling all waiters. Regression test drives _read_loop with _ws=None against a registered pending future. --- gateway/relay/ws_transport.py | 9 ++++++++- .../relay/test_ws_transport_hardening.py | 18 ++++++++++++++++++ 2 files changed, 26 insertions(+), 1 deletion(-) diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index e4924992c5..ae25006e9d 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -822,13 +822,20 @@ class WebSocketRelayTransport: await self._ws.send(json.dumps(frame) + "\n") async def _read_loop(self) -> None: - assert self._ws is not None # Bind the socket this reader serves: the finally below must only # clear _ws if it still points at THIS socket (a supervisor re-dial # may have already installed a fresh one by the time we unwind). ws = self._ws buf = "" try: + if ws is None: + # Scheduled without a socket (a lifecycle bug, not a normal + # path). The old `assert` here escaped BEFORE the finally + # existed to fail pending futures — the one exit that could + # still strand waiters for the full outbound timeout. Fall + # through to the finally instead; it settles them all. + logger.error("relay ws read loop started with no socket") + return try: async for chunk in self._ws: buf += chunk if isinstance(chunk, str) else chunk.decode("utf-8") diff --git a/tests/gateway/relay/test_ws_transport_hardening.py b/tests/gateway/relay/test_ws_transport_hardening.py index 88e5b151e1..4ead24a829 100644 --- a/tests/gateway/relay/test_ws_transport_hardening.py +++ b/tests/gateway/relay/test_ws_transport_hardening.py @@ -263,6 +263,24 @@ async def _run_reader_to_exit(t: WebSocketRelayTransport, fake: _DroppingWS) -> await t._reader +@pytest.mark.asyncio +async def test_read_loop_without_socket_still_fails_pending(): + """If the reader is ever scheduled with no socket (lifecycle bug), it must + still settle in-flight waiters on its way out — the old `assert` escaped + before the fail-pending cleanup and left them to the full 30s timeout.""" + t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=30.0) + loop = asyncio.get_running_loop() + fut: asyncio.Future = loop.create_future() + t._pending["rid"] = fut + t._ws = None + + await t._read_loop() # must not raise + + assert fut.done() + assert fut.result() == {"success": False, "error": "relay transport connection lost"} + assert t._pending == {} + + @pytest.mark.asyncio async def test_send_after_terminal_4401_revocation_fails_fast(): """A terminal 4401 revocation deliberately arms NO reconnect supervisor, From c60a525380d2d6f8cd3ee5fbc8e066922a11e073 Mon Sep 17 00:00:00 2001 From: Ben Barclay Date: Thu, 20 Aug 2026 15:26:31 +1000 Subject: [PATCH 10/42] fix(relay): dedupe key platform component is spelling-invariant MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The key derived its platform component with getattr(platform, 'value', ''), which handles the Platform enum the wire decoder always produces but collapses a plain-string platform — or a missing one — to the same empty string. Two DIFFERENT string platforms would then share one key component (cross-platform id collisions conflate), and enum vs string spellings of the SAME platform would produce two keys (a replay decoded differently would not dedupe). Inert on today's wire path (the decoder canonicalizes unknowns to Platform.RELAY), but alternate event constructors carry strings, so normalize at the key: enum value when present, the string itself otherwise, empty only when there is genuinely no platform. Missing platform intentionally still yields a key — fail-open on identity is reserved for missing message/chat ids. Tests pin all three properties; the spelling-invariance pair fails against the previous expression. --- gateway/relay/adapter.py | 9 +++++- .../relay/test_relay_inbound_dedupe.py | 29 +++++++++++++++++++ 2 files changed, 37 insertions(+), 1 deletion(-) diff --git a/gateway/relay/adapter.py b/gateway/relay/adapter.py index 2101a8faff..8139882373 100644 --- a/gateway/relay/adapter.py +++ b/gateway/relay/adapter.py @@ -414,7 +414,14 @@ class RelayAdapter(BasePlatformAdapter): chat_id = getattr(source, "chat_id", None) if not message_id or not chat_id: return None - platform = getattr(getattr(source, "platform", None), "value", "") + # Normalize the platform component: production wire decoding always + # yields a Platform enum (unknowns canonicalize to Platform.RELAY), + # but alternate constructors may carry the plain string. Use the + # enum's value when present, the string itself otherwise — both + # spellings of one platform must produce ONE key, and two different + # string platforms must not collapse into the same empty component. + raw_platform = getattr(source, "platform", None) + platform = getattr(raw_platform, "value", raw_platform) or "" return f"{platform}:{chat_id}:{message_id}" def _relay_slack_extra(self) -> Dict[str, Any]: diff --git a/tests/gateway/relay/test_relay_inbound_dedupe.py b/tests/gateway/relay/test_relay_inbound_dedupe.py index 0fd717fae1..008540a033 100644 --- a/tests/gateway/relay/test_relay_inbound_dedupe.py +++ b/tests/gateway/relay/test_relay_inbound_dedupe.py @@ -200,3 +200,32 @@ class TestWireLevelReplayDedupe: adapter._on_inbound(self._decode(source={"platform": "discord"})) ) assert len(handled) == 2 + + +class TestDedupeKeyPlatformNormalization: + """The platform component of the key must be spelling-invariant: a + Platform enum and its plain-string form are ONE platform (one key), and + two different string platforms must never collapse into a shared empty + component. Production wire decoding always yields the enum; alternate + event constructors may carry the string.""" + + def _key(self, platform): + adapter, _ = _connected_adapter() + source = SessionSource( + platform=platform, chat_id="C1", chat_type="channel", message_id="m1" + ) + event = MessageEvent(text="hi", source=source, message_id="m1") + return adapter._inbound_dedupe_key(event) + + def test_enum_and_string_spellings_produce_one_key(self): + assert self._key(Platform.SLACK) == self._key("slack") + + def test_distinct_string_platforms_stay_distinct(self): + assert self._key("slack") != self._key("discord") + + def test_missing_platform_still_yields_a_key(self): + # Fail-open on identity is reserved for missing message/chat ids; + # a missing platform alone must not disable dedupe. + key = self._key(None) + assert key is not None + assert key.startswith(":") From bb0a4193d1253758b77b16775ad3fc0b64cdc3a2 Mon Sep 17 00:00:00 2001 From: Ben Barclay Date: Thu, 20 Aug 2026 15:27:51 +1000 Subject: [PATCH 11/42] fix(relay): a raising socket write returns the result dict, not an exception MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The socket can die BETWEEN _request_response's 'is None' liveness guard and the actual write: the reader's finally hasn't cleared _ws yet, so _send raises ConnectionClosed straight into callers whose contract is a result dict (RelayAdapter.send consumes it with no try — only the cosmetic typing lanes wrap the call). No liveness check can close this window; it has to be caught at the write. Convert the raise to {'success': False, 'error': ...} like every other failed send, log the traceback at debug (the returned string alone can't distinguish an ordinary dead socket from a defect in the frame building above), and rely on the existing finally to drop the pending entry. CancelledError is a BaseException, so cancellation still propagates. Same disposition as the equivalent guard in PR #82238; regression test drives a socket whose write raises while the reader is still parked, and fails without the except clause. --- gateway/relay/ws_transport.py | 10 +++++ .../relay/test_ws_transport_hardening.py | 45 +++++++++++++++++++ 2 files changed, 55 insertions(+) diff --git a/gateway/relay/ws_transport.py b/gateway/relay/ws_transport.py index ae25006e9d..1018345900 100644 --- a/gateway/relay/ws_transport.py +++ b/gateway/relay/ws_transport.py @@ -812,6 +812,16 @@ class WebSocketRelayTransport: return await asyncio.wait_for(fut, timeout=self._outbound_timeout_s) except asyncio.TimeoutError: return {"success": False, "error": "relay outbound timed out"} + except Exception as exc: # noqa: BLE001 - a dead socket is a failed send, not a raise + # No `is None` check can close the window where the socket dies + # BETWEEN the liveness guard above and the actual write — the + # reader's finally hasn't cleared _ws yet, so _send raises + # ConnectionClosed straight into callers whose contract is a + # result dict (RelayAdapter.send consumes it with no try). + # Report it like every other failed send. CancelledError is a + # BaseException, so cancellation still propagates. + logger.debug("relay %s send failed", frame_type, exc_info=True) + return {"success": False, "error": f"relay send failed: {exc}"} finally: self._pending.pop(request_id, None) diff --git a/tests/gateway/relay/test_ws_transport_hardening.py b/tests/gateway/relay/test_ws_transport_hardening.py index 4ead24a829..af923af4bc 100644 --- a/tests/gateway/relay/test_ws_transport_hardening.py +++ b/tests/gateway/relay/test_ws_transport_hardening.py @@ -326,3 +326,48 @@ async def test_send_after_drop_with_reconnect_disabled_fails_fast(): ) assert result["success"] is False assert t._pending == {} + + +@pytest.mark.asyncio +async def test_send_raising_socket_returns_error_dict(): + """The socket can die BETWEEN the `_ws is None` liveness guard and the + actual write (the reader's finally hasn't cleared the handle yet). The + write then raises ConnectionClosed — but send_outbound's contract is a + result dict, and RelayAdapter.send consumes it with no try. The raise + must be converted to {"success": False, ...}, with no future left in + _pending.""" + + class _RaisingWS: + """Send raises (already dead); the reader hasn't noticed yet.""" + + def __init__(self): + self.reader_release = asyncio.Event() + + async def send(self, data): + raise ConnectionClosedError(None, None) + + def __aiter__(self): + return self + + async def __anext__(self): + await self.reader_release.wait() + raise ConnectionClosedError(None, None) + + async def close(self): + pass + + t = WebSocketRelayTransport("ws://unused", "discord", "bot1", outbound_timeout_s=5.0) + fake = _RaisingWS() + t._ws = fake + t._reader = asyncio.create_task(t._read_loop()) + await asyncio.sleep(0) + + result = await asyncio.wait_for( + t.send_outbound({"op": "send_message", "text": "hi"}), timeout=2.0 + ) + assert result["success"] is False + assert "relay send failed" in result["error"] + assert t._pending == {} + + fake.reader_release.set() + await t._reader From c25aa35e1aab21e0517574026b080e6087c0050d Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Wed, 19 Aug 2026 19:16:44 -0600 Subject: [PATCH 12/42] fix(desktop): keep peer windows on shared gateway --- apps/desktop/electron/main.ts | 16 ++++++++++--- apps/desktop/electron/session-windows.test.ts | 14 +++++++++++ apps/desktop/electron/session-windows.ts | 18 +++++++++++++++ .../chat/sidebar/connection-switcher.test.tsx | 23 +++++++++++++++++++ .../app/chat/sidebar/connection-switcher.tsx | 7 ++++-- apps/desktop/src/store/windows.test.ts | 17 +++++++++++++- apps/desktop/src/store/windows.ts | 12 ++++++++++ 7 files changed, 101 insertions(+), 6 deletions(-) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index 85bec53638..4eb6d039c7 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -256,6 +256,7 @@ import { import { missingRendererAssets } from './renderer-bundle' import { attachRendererConsoleCapture, formatRendererBoundaryReport } from './renderer-log' import { + buildInstanceWindowUrl, buildSessionWindowUrl, chatWindowWebPreferences, createSessionWindowRegistry, @@ -10870,8 +10871,10 @@ function createSessionWindow(sessionId, { watch = false } = {}) { // Additional full "instance" windows — peers of the primary that render the // COMPLETE app (sidebar, routing, its own draft) against the shared backend, so // a user can run multiple GUI windows at once (⌘⇧N / the "New Window" palette -// command). Unlike the compact session windows they carry no `?win` flag. The -// primary mainWindow stays the notification / deep-link / pet-overlay anchor and +// command). Unlike the compact session windows they carry no `?win` flag; a +// separate `peer=1` marker prevents them from replaying app-launch source +// restoration after joining that shared backend. The primary mainWindow stays +// the notification / deep-link / pet-overlay anchor and // is NOT tracked here. The set holds a strong reference so an open peer isn't // garbage-collected, and drops it on close. const instanceWindows = new Set() @@ -10951,7 +10954,14 @@ function createInstanceWindow() { }) attachRendererConsoleCapture(win, 'instance', rememberLog) - loadWindowUrl(win, DEV_SERVER || pathToFileURL(resolveRendererIndex()).toString(), 'Instance window') + loadWindowUrl( + win, + buildInstanceWindowUrl({ + devServer: DEV_SERVER, + rendererIndexPath: DEV_SERVER ? undefined : resolveRendererIndex() + }), + 'Instance window' + ) return win } diff --git a/apps/desktop/electron/session-windows.test.ts b/apps/desktop/electron/session-windows.test.ts index 1959cc4458..c0a1178429 100644 --- a/apps/desktop/electron/session-windows.test.ts +++ b/apps/desktop/electron/session-windows.test.ts @@ -3,6 +3,7 @@ import assert from 'node:assert/strict' import { test } from 'vitest' import { + buildInstanceWindowUrl, buildSessionWindowUrl, chatWindowWebPreferences, createSessionWindowRegistry, @@ -88,6 +89,19 @@ test('buildSessionWindowUrl adds the watch flag for spectator windows, before th assert.equal(url, 'http://localhost:5173/?win=secondary&watch=1#/abc') }) +test('buildInstanceWindowUrl marks a full peer without selecting a specialized renderer', () => { + const url = buildInstanceWindowUrl({ devServer: 'http://localhost:5173/' }) + + assert.equal(url, 'http://localhost:5173/?peer=1') + assert.ok(!url.includes('win=')) +}) + +test('buildInstanceWindowUrl marks a packaged full peer', () => { + const url = buildInstanceWindowUrl({ rendererIndexPath: '/opt/app/index.html' }) + + assert.match(url, /^file:\/\/.*index\.html\?peer=1$/) +}) + test('instanceWindowBounds cascades a new window off its source bounds', () => { const bounds = instanceWindowBounds({ x: 100, y: 120, width: 1400, height: 900 }, { width: 1, height: 1 }) diff --git a/apps/desktop/electron/session-windows.ts b/apps/desktop/electron/session-windows.ts index 790f1c03a7..19be87ff1c 100644 --- a/apps/desktop/electron/session-windows.ts +++ b/apps/desktop/electron/session-windows.ts @@ -77,6 +77,23 @@ function buildSessionWindowUrl(sessionId: string, { devServer, rendererIndexPath return `${pathToFileURL(rendererIndexPath).toString()}${query}${route}` } +// Full peer windows render the ordinary app shell, so they deliberately do +// not use the `win` query parameter that selects a specialized renderer. The +// separate marker lets the renderer distinguish a peer from the one primary +// app window: app-launch source restoration belongs to the primary only, while +// a peer keeps the already-running backend it joined during boot. +function buildInstanceWindowUrl({ devServer, rendererIndexPath }: any = {}) { + const query = '?peer=1' + + if (devServer) { + const base = devServer.endsWith('/') ? devServer.slice(0, -1) : devServer + + return `${base}/${query}` + } + + return `${pathToFileURL(rendererIndexPath).toString()}${query}` +} + // Full "instance" windows (⌘⇧N / the "New Window" command) open a complete app // peer, not a compact chat. Cascade each one off its source window's bounds so a // new window doesn't land exactly on top of the one it was spawned from. Pure so @@ -160,6 +177,7 @@ function createSessionWindowRegistry() { } export { + buildInstanceWindowUrl, buildSessionWindowUrl, chatWindowWebPreferences, createSessionWindowRegistry, diff --git a/apps/desktop/src/app/chat/sidebar/connection-switcher.test.tsx b/apps/desktop/src/app/chat/sidebar/connection-switcher.test.tsx index fe9ef7afd5..79acb22e86 100644 --- a/apps/desktop/src/app/chat/sidebar/connection-switcher.test.tsx +++ b/apps/desktop/src/app/chat/sidebar/connection-switcher.test.tsx @@ -40,6 +40,10 @@ vi.mock('@/store/boot', () => ({ }) })) +vi.mock('@/store/windows', () => ({ + isPeerInstanceWindow: vi.fn(() => false) +})) + vi.mock('@/i18n', () => ({ useI18n: () => ({ t: { @@ -65,6 +69,7 @@ vi.mock('@/i18n', () => ({ const connectionStore = await import('@/store/connections') const bootStore = await import('@/store/boot') +const windowStore = await import('@/store/windows') const $activeConnectionId = connectionStore.$activeConnectionId as ReturnType> const $connectionsRegistry = connectionStore.$connectionsRegistry const $desktopBoot = bootStore.$desktopBoot @@ -72,6 +77,7 @@ const $pendingConnectionId = connectionStore.$pendingConnectionId const initializeConnectionsRegistry = vi.mocked(connectionStore.initializeConnectionsRegistry) const refreshConnectionsRegistry = vi.mocked(connectionStore.refreshConnectionsRegistry) const selectConnection = vi.mocked(connectionStore.selectConnection) +const isPeerInstanceWindow = vi.mocked(windowStore.isPeerInstanceWindow) const onConnect = vi.fn() const connection = (id: string, label: string, kind: 'local' | 'remote' = 'remote') => ({ @@ -106,6 +112,7 @@ afterEach(() => { }) $pendingConnectionId.set(null) $findInPage.set({ active: false, query: '', matchOrdinal: 0, matchCount: 0 }) + isPeerInstanceWindow.mockReturnValue(false) }) describe('ConnectionSwitcher', () => { @@ -127,6 +134,22 @@ describe('ConnectionSwitcher', () => { await waitFor(() => expect(initializeConnectionsRegistry).toHaveBeenCalledTimes(1)) }) + it('keeps a full peer on the shared backend instead of replaying app-launch source restoration', async () => { + isPeerInstanceWindow.mockReturnValue(true) + $desktopBoot.set({ + ...$desktopBoot.get(), + phase: 'renderer.ready', + progress: 100, + running: false, + visible: false + }) + + render() + + await waitFor(() => expect(refreshConnectionsRegistry).toHaveBeenCalledTimes(1)) + expect(initializeConnectionsRegistry).not.toHaveBeenCalled() + }) + it('adds no source chrome for a local-only setup', () => { $connectionsRegistry.set(registry([connection('local', 'This device', 'local')])) render() diff --git a/apps/desktop/src/app/chat/sidebar/connection-switcher.tsx b/apps/desktop/src/app/chat/sidebar/connection-switcher.tsx index f0396abfd2..22f559e7e3 100644 --- a/apps/desktop/src/app/chat/sidebar/connection-switcher.tsx +++ b/apps/desktop/src/app/chat/sidebar/connection-switcher.tsx @@ -36,6 +36,7 @@ import { } from '@/store/connections' import { closeFindBar } from '@/store/find-in-page' import { notifyError } from '@/store/notifications' +import { isPeerInstanceWindow } from '@/store/windows' export function ConnectionSwitcher({ compact = false, onConnect }: { compact?: boolean; onConnect: () => void }) { const { t } = useI18n() @@ -64,8 +65,10 @@ export function ConnectionSwitcher({ compact = false, onConnect }: { compact?: b // The primary boot owns its initial config/session fetches. Restoring a // different source before those settle lets a late primary response repaint // the sidebar under the new source label. Switch only after boot completes, - // then the normal source reset/refetch remains the final writer. - if (!boot.running) { + // then the normal source reset/refetch remains the final writer. Full peer + // windows already joined the shared backend during their own boot; replaying + // this app-launch preference there would move them away from that backend. + if (!boot.running && !isPeerInstanceWindow()) { void initializeConnectionsRegistry().catch(() => undefined) } }, [boot.running]) diff --git a/apps/desktop/src/store/windows.test.ts b/apps/desktop/src/store/windows.test.ts index e6c3f4974a..d6b745be85 100644 --- a/apps/desktop/src/store/windows.test.ts +++ b/apps/desktop/src/store/windows.test.ts @@ -1,6 +1,12 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' -import { canOpenNewWindow, canOpenSessionWindow, openNewWindow, openSessionInNewWindow } from './windows' +import { + canOpenNewWindow, + canOpenSessionWindow, + isPeerInstanceWindow, + openNewWindow, + openSessionInNewWindow +} from './windows' const desktopWindow = window as unknown as { hermesDesktop?: Window['hermesDesktop'] } const initialHermesDesktop = desktopWindow.hermesDesktop @@ -50,6 +56,15 @@ describe('canOpenSessionWindow', () => { }) }) +describe('isPeerInstanceWindow', () => { + it('recognizes only the full peer marker', () => { + expect(isPeerInstanceWindow('?peer=1')).toBe(true) + expect(isPeerInstanceWindow('?peer=0')).toBe(false) + expect(isPeerInstanceWindow('?win=secondary')).toBe(false) + expect(isPeerInstanceWindow('')).toBe(false) + }) +}) + describe('openSessionInNewWindow', () => { it('no-ops without a session id', async () => { const open = vi.fn().mockResolvedValue({ ok: true }) diff --git a/apps/desktop/src/store/windows.ts b/apps/desktop/src/store/windows.ts index 3549ce4186..1403863d53 100644 --- a/apps/desktop/src/store/windows.ts +++ b/apps/desktop/src/store/windows.ts @@ -83,6 +83,18 @@ export function isWatchWindow(): boolean { // keystroke into N prompts, and a HUD is the last place to paint onboarding. export const isAuxiliaryWindow = (): boolean => isSecondaryWindow() || isHudWindow() +// A full peer window renders the ordinary app shell against the backend that +// Electron already has running. It is not an auxiliary/specialized renderer, +// but it must not replay the primary window's app-launch source restoration +// after boot and silently re-home itself to another registered gateway. +export function isPeerInstanceWindow(search = typeof window === 'undefined' ? '' : window.location.search): boolean { + try { + return new URLSearchParams(search).get('peer') === '1' + } catch { + return false + } +} + // The profile a helper window (the HUD) was asked to boot against, carried in // the query string by the main process (see hudUrl). The HUD is a full app // renderer that otherwise adopts the PRIMARY backend's profile — wrong the From 9ae7247616966a02428f2442c414f14d9cdbce12 Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Wed, 19 Aug 2026 23:44:41 -0600 Subject: [PATCH 13/42] fix(desktop): skip source restore in auxiliary windows --- .../chat/sidebar/connection-switcher.test.tsx | 19 +++++++++++++++++++ .../app/chat/sidebar/connection-switcher.tsx | 10 +++++----- 2 files changed, 24 insertions(+), 5 deletions(-) diff --git a/apps/desktop/src/app/chat/sidebar/connection-switcher.test.tsx b/apps/desktop/src/app/chat/sidebar/connection-switcher.test.tsx index 79acb22e86..fa0af2324f 100644 --- a/apps/desktop/src/app/chat/sidebar/connection-switcher.test.tsx +++ b/apps/desktop/src/app/chat/sidebar/connection-switcher.test.tsx @@ -41,6 +41,7 @@ vi.mock('@/store/boot', () => ({ })) vi.mock('@/store/windows', () => ({ + isAuxiliaryWindow: vi.fn(() => false), isPeerInstanceWindow: vi.fn(() => false) })) @@ -77,6 +78,7 @@ const $pendingConnectionId = connectionStore.$pendingConnectionId const initializeConnectionsRegistry = vi.mocked(connectionStore.initializeConnectionsRegistry) const refreshConnectionsRegistry = vi.mocked(connectionStore.refreshConnectionsRegistry) const selectConnection = vi.mocked(connectionStore.selectConnection) +const isAuxiliaryWindow = vi.mocked(windowStore.isAuxiliaryWindow) const isPeerInstanceWindow = vi.mocked(windowStore.isPeerInstanceWindow) const onConnect = vi.fn() @@ -112,6 +114,7 @@ afterEach(() => { }) $pendingConnectionId.set(null) $findInPage.set({ active: false, query: '', matchOrdinal: 0, matchCount: 0 }) + isAuxiliaryWindow.mockReturnValue(false) isPeerInstanceWindow.mockReturnValue(false) }) @@ -150,6 +153,22 @@ describe('ConnectionSwitcher', () => { expect(initializeConnectionsRegistry).not.toHaveBeenCalled() }) + it('keeps a secondary session window from replaying app-launch source restoration', async () => { + isAuxiliaryWindow.mockReturnValue(true) + $desktopBoot.set({ + ...$desktopBoot.get(), + phase: 'renderer.ready', + progress: 100, + running: false, + visible: false + }) + + render() + + await waitFor(() => expect(refreshConnectionsRegistry).toHaveBeenCalledTimes(1)) + expect(initializeConnectionsRegistry).not.toHaveBeenCalled() + }) + it('adds no source chrome for a local-only setup', () => { $connectionsRegistry.set(registry([connection('local', 'This device', 'local')])) render() diff --git a/apps/desktop/src/app/chat/sidebar/connection-switcher.tsx b/apps/desktop/src/app/chat/sidebar/connection-switcher.tsx index 22f559e7e3..ce0e56b801 100644 --- a/apps/desktop/src/app/chat/sidebar/connection-switcher.tsx +++ b/apps/desktop/src/app/chat/sidebar/connection-switcher.tsx @@ -36,7 +36,7 @@ import { } from '@/store/connections' import { closeFindBar } from '@/store/find-in-page' import { notifyError } from '@/store/notifications' -import { isPeerInstanceWindow } from '@/store/windows' +import { isAuxiliaryWindow, isPeerInstanceWindow } from '@/store/windows' export function ConnectionSwitcher({ compact = false, onConnect }: { compact?: boolean; onConnect: () => void }) { const { t } = useI18n() @@ -65,10 +65,10 @@ export function ConnectionSwitcher({ compact = false, onConnect }: { compact?: b // The primary boot owns its initial config/session fetches. Restoring a // different source before those settle lets a late primary response repaint // the sidebar under the new source label. Switch only after boot completes, - // then the normal source reset/refetch remains the final writer. Full peer - // windows already joined the shared backend during their own boot; replaying - // this app-launch preference there would move them away from that backend. - if (!boot.running && !isPeerInstanceWindow()) { + // then the normal source reset/refetch remains the final writer. Peer and + // auxiliary windows already boot into their intended runtime; replaying the + // primary window's app-launch preference would move them away from it. + if (!boot.running && !isAuxiliaryWindow() && !isPeerInstanceWindow()) { void initializeConnectionsRegistry().catch(() => undefined) } }, [boot.running]) From 62016a1b0a4099ad20ce6c09e7b2417a1bac78e6 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Tue, 18 Aug 2026 03:11:40 +0800 Subject: [PATCH 14/42] fix(compression): count a would-grow refusal as an ineffective strike The anti-growth guard correctly refuses to persist a compressed candidate larger than the original, but the rejection was never recorded by the anti-thrashing breaker: _ineffective_compression_count stayed at zero, the latch never tripped, and automatic compression retried the SAME unchanged transcript on every turn - same summary request, same refusal, same user-facing warning (#88568). Add ContextCompressor.record_rejected_compaction(): one persisted ineffective strike, without arming post-compaction real-usage verification (nothing was committed) and without touching the fallback-summary streak (no summary was accepted). The would-grow abort path in conversation_compression calls it before returning the original transcript. Two refusals latch the normal breaker, manual /compress keeps bypassing it (force=True), and the existing recovery window still allows one probe later. Fixes #88568 --- agent/context_compressor.py | 25 ++++++++++++ agent/conversation_compression.py | 14 +++++++ tests/agent/test_compaction_anti_thrash.py | 44 ++++++++++++++++++++++ 3 files changed, 83 insertions(+) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 03639cf985..8c33b9be6c 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -2373,6 +2373,31 @@ class ContextCompressor(ContextEngine): self._ineffective_compression_count = count self._persist_ineffective_compression_count() + def record_rejected_compaction(self) -> None: + """Record one compaction whose result was REJECTED before committing. + + The anti-growth guard in the commit layer (conversation_compression) + discards a candidate that would grow the transcript and keeps the + original. Without recording the attempt, the anti-thrash breaker + never sees a strike, so automatic compression retries the SAME + unchanged transcript on every turn — same summary request, same + refusal, same user-facing warning (#88568). This counts one + ineffective strike (persisted, so the normal >= 2 latch and its + recovery window apply) WITHOUT arming post-compaction real-usage + verification — nothing was committed, so there is no new compaction + to verify — and without touching the fallback-summary streak (no + summary was accepted). + """ + self._record_ineffective_compression_verdict( + self._ineffective_compression_count + 1 + ) + if not self.quiet_mode: + logger.warning( + "Compaction rejected before commit (would grow the " + "transcript); ineffective_compression_count=%d", + self._ineffective_compression_count, + ) + def record_completed_compaction( self, *, used_fallback: bool = False, feasibility_skip: bool = False, ) -> None: diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index ba142416b3..77a8a4ab93 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -3425,6 +3425,20 @@ def compress_context( split_status="aborted", failure_class="would_grow", ) + # Record the rejected attempt as an ineffective + # compaction strike so the anti-thrash breaker latches + # after the normal threshold. Without this, the unchanged + # transcript stays over the compression threshold and + # automatic compression retries the identical summary + # request on every turn (#88568). Manual /compress keeps + # bypassing the latch (force=True skips the guards). + try: + agent.context_compressor.record_rejected_compaction() + except Exception: + logger.debug( + "could not record rejected-compaction strike", + exc_info=True, + ) _release_lock() return messages, _existing_sp diff --git a/tests/agent/test_compaction_anti_thrash.py b/tests/agent/test_compaction_anti_thrash.py index a862668158..457f28a5c3 100644 --- a/tests/agent/test_compaction_anti_thrash.py +++ b/tests/agent/test_compaction_anti_thrash.py @@ -203,3 +203,47 @@ class TestMinimumMessagesBranch: assert len(out) == len(msgs), "nothing should have been compressed" assert cc._last_compression_made_progress is False assert cc._ineffective_compression_count == before + 1 + + +class TestRejectedCompactionStrike: + """#88568 — a would-grow refusal must count as an ineffective strike. + + The anti-growth guard correctly keeps the original transcript, but the + rejection used to leave ``_ineffective_compression_count`` untouched, so + the breaker never latched and automatic compression retried the SAME + unchanged transcript on every turn. + """ + + def test_rejected_compaction_increments_strike(self): + cc = _compressor(threshold_tokens=1) + assert cc._ineffective_compression_count == 0 + + cc.record_rejected_compaction() + + assert cc._ineffective_compression_count == 1 + + def test_two_rejections_stop_further_automatic_compression(self): + cc = _compressor(threshold_tokens=1) + cc.record_rejected_compaction() + cc.record_rejected_compaction() + + # The latch consumers key on the counter itself (>= 2 blocks); + # pin the counter and the recovery-clock arming side effect. + assert cc._ineffective_compression_count >= 2 + + def test_rejection_does_not_arm_real_usage_verification(self): + """Nothing was committed, so the next response must not be scored + against the pre-rejection transcript (that verdict belongs to + committed compactions only).""" + cc = _compressor(threshold_tokens=1) + cc.record_rejected_compaction() + + assert cc._verify_compaction_cleared_threshold is False + + def test_rejection_leaves_fallback_streak_untouched(self): + cc = _compressor(threshold_tokens=1) + cc._fallback_compression_streak = 1 + + cc.record_rejected_compaction() + + assert cc._fallback_compression_streak == 1 From fb96247eaf1f7ee0415a380de590c88cbf7a3397 Mon Sep 17 00:00:00 2001 From: MindDragonLabs <314294625+MindDragonLabs@users.noreply.github.com> Date: Thu, 20 Aug 2026 10:52:45 +0530 Subject: [PATCH 15/42] fix(compression): salvage grown candidates before refusal --- agent/context_compressor.py | 96 +++++++++++++++ agent/conversation_compression.py | 19 +++ tests/agent/test_salvage_grown_transcript.py | 119 +++++++++++++++++++ tests/run_agent/test_in_place_compaction.py | 54 +++++++++ 4 files changed, 288 insertions(+) create mode 100644 tests/agent/test_salvage_grown_transcript.py diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 8c33b9be6c..86baf5f1f3 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -350,6 +350,102 @@ _SUMMARY_END_MARKER = ( _MERGED_PRIOR_CONTEXT_HEADER = "[PRIOR CONTEXT — for reference only; not a new message]" _MERGED_SUMMARY_DELIMITER = "[END OF PRIOR CONTEXT — COMPACTION SUMMARY BELOW]" +_SALVAGE_TOOL_PLACEHOLDER = "[Old tool output cleared to save context space]" +_SALVAGE_SUMMARY_MAX_CHARS = 8_000 +_SALVAGE_KEEP_RECENT_TOOLS = 2 +_SALVAGE_REASONING_KEYS = ("reasoning", "reasoning_content", "reasoning_details") + + +def _looks_like_compaction_summary(msg: Dict[str, Any], content: str) -> bool: + # Only cap a standalone handoff. Merged carriers preserve a real tail ask + # in the same content string; truncating those could delete live user text. + if not content.rstrip().endswith(_SUMMARY_END_MARKER): + return False + if content.startswith(_MERGED_PRIOR_CONTEXT_HEADER): + return False + # Content heuristics alone must never authorize mutating a live user turn. + # Compressor-generated user-role summaries carry this private marker; + # ordinary user input does not. + if ( + msg.get("role") == "user" + and not msg.get(COMPRESSED_SUMMARY_METADATA_KEY) + ): + return False + head = content[:280] + return ( + bool(msg.get(COMPRESSED_SUMMARY_METADATA_KEY)) + or "CONTEXT COMPACTION" in head + or "[CONTEXT COMPACTION]" in head + or "Conversation Summary" in head + ) + + +def salvage_grown_transcript( + original: List[Dict[str, Any]], + candidate: List[Dict[str, Any]], +) -> Optional[List[Dict[str, Any]]]: + """Mechanically shrink a compression candidate, or return ``None``. + + Already-compacted middles can be summarized slightly larger while retained + tool bodies, stale reasoning, or a synthetic todo snapshot tip the final + candidate over the input size. Work on copies and admit the salvage only + when the same rough estimator proves it is strictly smaller than the input. + """ + if not candidate or not original: + return None + budget = estimate_messages_tokens_rough(original) + if budget <= 0: + return None + + out: List[Dict[str, Any]] = [] + tool_indices: List[int] = [] + last_assistant_idx = -1 + for msg in candidate: + if not isinstance(msg, dict): + out.append(msg) + continue + if msg.get("_todo_snapshot_synthetic") and msg.get("role") == "user": + continue + copied = dict(msg) + out.append(copied) + role = copied.get("role") + if role == "tool": + tool_indices.append(len(out) - 1) + elif role == "assistant": + last_assistant_idx = len(out) - 1 + + keep_tools = set(tool_indices[-_SALVAGE_KEEP_RECENT_TOOLS:]) + for index, msg in enumerate(out): + if not isinstance(msg, dict): + continue + if msg.get("role") == "assistant" and index != last_assistant_idx: + for key in _SALVAGE_REASONING_KEYS: + msg.pop(key, None) + if msg.get("role") == "tool" and index not in keep_tools: + content = msg.get("content") + if isinstance(content, str) and len(content) > 200: + msg["content"] = _SALVAGE_TOOL_PLACEHOLDER + content = msg.get("content") + if ( + isinstance(content, str) + and len(content) > _SALVAGE_SUMMARY_MAX_CHARS + and _looks_like_compaction_summary(msg, content) + ): + msg["content"] = ( + content[:_SALVAGE_SUMMARY_MAX_CHARS].rstrip() + + "\n…[summary truncated so compaction can shrink]\n\n" + + _SUMMARY_END_MARKER + ) + + if not any( + isinstance(message, dict) and message.get("role") == "user" + for message in out + ): + return None + if estimate_messages_tokens_rough(out) < budget: + return out + return None + # Handoff prefixes that shipped in earlier releases. A summary persisted under # one of these can be inherited into a resumed lineage (#35344); when it is # re-normalized on re-compaction we must strip the OLD prefix too, otherwise the diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 77a8a4ab93..521c08d928 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -3386,6 +3386,25 @@ def compress_context( # transcript stays untouched and durable. _rough_in = estimate_messages_tokens_rough(messages) _rough_out = estimate_messages_tokens_rough(compressed) + if _rough_out > _rough_in: + # Todo refresh and user-turn anchoring happen after the + # compressor's own size check, so they can tip a break-even + # candidate over. Give it one mechanical salvage pass. + from agent.context_compressor import salvage_grown_transcript + + _salvaged = salvage_grown_transcript(messages, compressed) + if _salvaged is not None: + _salv_est = estimate_messages_tokens_rough(_salvaged) + if _salv_est < _rough_in: + logger.info( + "Compression salvage recovered a shrinking " + "transcript (session=%s, ~%s -> ~%s tokens)", + agent.session_id or "none", + f"{_rough_in:,}", + f"{_salv_est:,}", + ) + compressed = _salvaged + _rough_out = _salv_est if _rough_out > _rough_in: logger.warning( "Compression refused: compressed transcript would be " diff --git a/tests/agent/test_salvage_grown_transcript.py b/tests/agent/test_salvage_grown_transcript.py new file mode 100644 index 0000000000..920dd85a1c --- /dev/null +++ b/tests/agent/test_salvage_grown_transcript.py @@ -0,0 +1,119 @@ +"""Mechanical salvage for compression candidates that would grow.""" + +from agent.context_compressor import ( + COMPRESSED_SUMMARY_METADATA_KEY, + _SUMMARY_END_MARKER, + salvage_grown_transcript, +) +from agent.model_metadata import estimate_messages_tokens_rough + + +def test_salvage_stubs_old_tools_and_drops_todo(): + original = [ + {"role": "user", "content": "go"}, + {"role": "assistant", "content": "ok"}, + {"role": "tool", "tool_call_id": "a", "content": "A" * 4000}, + {"role": "tool", "tool_call_id": "b", "content": "B" * 4000}, + {"role": "tool", "tool_call_id": "c", "content": "keep-latest"}, + ] + grown = original + [ + { + "role": "user", + "content": "Current todos:\n- [ ] x", + "_todo_snapshot_synthetic": True, + } + ] + assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original) + + out = salvage_grown_transcript(original, grown) + + assert out is not None + assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original) + assert not any(m.get("_todo_snapshot_synthetic") for m in out) + tools = [m["content"] for m in out if m.get("role") == "tool"] + assert tools[-1] == "keep-latest" + assert any("cleared to save context space" in t for t in tools) + + +def test_salvage_returns_none_when_nothing_can_shrink(): + original = [{"role": "user", "content": "tiny"}] + huge = [{"role": "user", "content": "X" * 200_000}] + + assert salvage_grown_transcript(original, huge) is None + + +def test_salvage_caps_oversized_summary(): + original = [ + {"role": "user", "content": "ask " + ("o" * 12_000)}, + {"role": "assistant", "content": "short reply"}, + ] + grown = [ + { + "role": "user", + "content": ( + "[CONTEXT COMPACTION] " + + ("S" * 20_000) + + "\n\n" + + _SUMMARY_END_MARKER + ), + COMPRESSED_SUMMARY_METADATA_KEY: True, + }, + {"role": "assistant", "content": "short reply"}, + ] + assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original) + + out = salvage_grown_transcript(original, grown) + + assert out is not None + assert len(out[0]["content"]) < 12_000 + assert "truncated so compaction can shrink" in out[0]["content"] + assert out[0]["content"].endswith(_SUMMARY_END_MARKER) + + +def test_salvage_never_truncates_merged_summary_with_live_user_tail(): + original = [{"role": "user", "content": "O" * 12_000}] + merged = [ + { + "role": "user", + "content": ( + "[CONTEXT COMPACTION] " + + ("S" * 20_000) + + "\n\n" + + _SUMMARY_END_MARKER + + "\n\nLIVE USER REQUEST" + ), + COMPRESSED_SUMMARY_METADATA_KEY: True, + } + ] + + assert salvage_grown_transcript(original, merged) is None + assert merged[0]["content"].endswith("LIVE USER REQUEST") + + +def test_salvage_does_not_cap_plain_user_text_quoting_summary_marker(): + original = [{"role": "user", "content": "O" * 20_000}] + quoted = "ordinary user text " + ("Q" * 12_000) + "\n\n" + _SUMMARY_END_MARKER + candidate = [{"role": "user", "content": quoted}] + + out = salvage_grown_transcript(original, candidate) + + assert out is not None + assert out[0]["content"] == quoted + assert "truncated so compaction can shrink" not in out[0]["content"] + + +def test_salvage_never_caps_unmarked_summary_shaped_live_user_text(): + original = [{"role": "user", "content": "O" * 20_000}] + live_user_text = ( + "[CONTEXT COMPACTION] " + + ("U" * 12_000) + + "\n\n" + + _SUMMARY_END_MARKER + ) + candidate = [{"role": "user", "content": live_user_text}] + + out = salvage_grown_transcript(original, candidate) + + assert out is not None + assert out[0]["content"] == live_user_text + assert "truncated so compaction can shrink" not in out[0]["content"] diff --git a/tests/run_agent/test_in_place_compaction.py b/tests/run_agent/test_in_place_compaction.py index 21b41f19cd..e78e63ce52 100644 --- a/tests/run_agent/test_in_place_compaction.py +++ b/tests/run_agent/test_in_place_compaction.py @@ -293,6 +293,60 @@ class TestInPlaceAntiGrowthGuard: assert agent.session_id == sid assert db.get_session(sid)["end_reason"] is None + def test_in_place_salvages_near_break_even_growth(self): + """Fat retained tool output + todo state should be salvaged and committed.""" + from hermes_state import SessionDB + from agent.conversation_compression import compress_context + from agent.model_metadata import estimate_messages_tokens_rough + + with tempfile.TemporaryDirectory() as tmp: + db = SessionDB(db_path=Path(tmp) / "t.db") + sid = "20260819_salvage" + _seed(db, sid, "salvage") + agent = _make_agent(db, sid, in_place=True) + agent._last_flushed_db_idx = 5 + + tool_calls = [ + {"id": call_id, "function": {"name": "terminal", "arguments": "{}"}} + for call_id in ("c1", "c2", "c3") + ] + original = [ + {"role": "user", "content": "do the work"}, + {"role": "assistant", "content": "calling tools", "tool_calls": tool_calls}, + {"role": "tool", "tool_call_id": "c1", "content": "OLD1 " + ("x" * 8000)}, + {"role": "tool", "tool_call_id": "c2", "content": "OLD2 " + ("y" * 8000)}, + {"role": "tool", "tool_call_id": "c3", "content": "keep-me " + ("z" * 100)}, + {"role": "user", "content": "thanks"}, + ] + grown = [ + {"role": "user", "content": "[CONTEXT COMPACTION] " + ("S" * 200)}, + {"role": "assistant", "content": "calling tools", "tool_calls": tool_calls}, + {"role": "tool", "tool_call_id": "c1", "content": "OLD1 " + ("x" * 8000)}, + {"role": "tool", "tool_call_id": "c2", "content": "OLD2 " + ("y" * 8000)}, + {"role": "tool", "tool_call_id": "c3", "content": "keep-me " + ("z" * 100)}, + { + "role": "user", + "content": "Current todos:\n- [ ] leftover", + "_todo_snapshot_synthetic": True, + }, + ] + assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original) + compressor = getattr(agent, "context_compressor") + compressor.compress = ( + lambda messages, current_tokens=None, focus_topic=None, force=False: grown + ) + + compressed, _sp = compress_context( + agent, original, approx_tokens=100_000, system_message="sys" + ) + + assert getattr(agent, "_last_compaction_in_place") is True + assert estimate_messages_tokens_rough(compressed) < estimate_messages_tokens_rough(original) + tool_bodies = [m.get("content") for m in compressed if m.get("role") == "tool"] + assert any(isinstance(body, str) and body.startswith("keep-me") for body in tool_bodies) + assert any("cleared to save context space" in (body or "") for body in tool_bodies) + assert not any(m.get("_todo_snapshot_synthetic") for m in compressed) + def test_in_place_still_commits_shrinking_compression(self): """The guard must not block legitimate compressions — a result SMALLER than the input still commits in place (regression net for #83339).""" From 5c03fbedc6ea19937fecb701f70c1644d09a1650 Mon Sep 17 00:00:00 2001 From: MindDragonLabs <314294625+MindDragonLabs@users.noreply.github.com> Date: Thu, 20 Aug 2026 10:52:45 +0530 Subject: [PATCH 16/42] fix(config): recognize memory nudge interval --- cli-config.yaml.example | 5 ----- hermes_cli/config_defaults.py | 4 ++++ tests/hermes_cli/test_set_config_value.py | 7 +++++++ 3 files changed, 11 insertions(+), 5 deletions(-) diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 87aad0ae94..472b7b98c3 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -833,11 +833,6 @@ memory: # every N user turns. Set to 0 to disable. Only active when memory is enabled. nudge_interval: 10 # Nudge every 10 user turns (0 = disabled) - # Memory flush: give the agent one turn to save memories before context is - # lost (compression, /new, /reset, exit). Set to 0 to disable. - # For exit/reset, only fires if the session had at least this many user turns. - flush_min_turns: 6 # Min user turns to trigger flush on exit/reset (0 = disabled) - # ============================================================================= # Session Reset Policy (Messaging Platforms) # ============================================================================= diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index cccb4b638f..bd99919cd2 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1865,6 +1865,10 @@ DEFAULT_CONFIG = { "write_approval": False, "memory_char_limit": 2200, # ~800 tokens at 2.75 chars/token "user_char_limit": 1375, # ~500 tokens at 2.75 chars/token + # Periodic built-in memory review. External providers with automatic + # turn/session extraction can set this to 0 and keep the small local + # store reserved for explicit high-frequency operational facts. + "nudge_interval": 10, # External memory provider plugin (empty = built-in only). # Set to a provider name to activate: "openviking", "mem0", # "hindsight", "holographic", "retaindb", "byterover". diff --git a/tests/hermes_cli/test_set_config_value.py b/tests/hermes_cli/test_set_config_value.py index e764890d3c..cd62cda456 100644 --- a/tests/hermes_cli/test_set_config_value.py +++ b/tests/hermes_cli/test_set_config_value.py @@ -111,6 +111,13 @@ class TestConfigYamlRouting: assert "not a recognized config key" not in capsys.readouterr().out assert "script_timeout_seconds: 600" in _read_config(_isolated_hermes_home) + def test_memory_nudge_interval_is_recognized(self, _isolated_hermes_home, capsys): + """The documented background-memory review interval is runtime config.""" + set_config_value("memory.nudge_interval", "0") + + assert "not a recognized config key" not in capsys.readouterr().out + assert "nudge_interval: 0" in _read_config(_isolated_hermes_home) + def test_terminal_docker_cwd_mount_flag_goes_to_config_and_env(self, _isolated_hermes_home): set_config_value("terminal.docker_mount_cwd_to_workspace", "true") config = _read_config(_isolated_hermes_home) From 0596ccdeb3b9b158fd77f7ef0b3c65ac53de4697 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 20 Aug 2026 11:04:39 +0530 Subject: [PATCH 17/42] =?UTF-8?q?fix(compression):=20salvage=20follow-up?= =?UTF-8?q?=20=E2=80=94=20todo=20snapshot=20last-resort,=20reuse=20prune?= =?UTF-8?q?=20helpers?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-up on the salvaged #90353: - Todo snapshot (+ coupled pruned-skill reload notice, 7a16840add) is now reduced only as a LAST resort after reasoning/tool/summary shrink ops, and the reload notice survives even then. - Reuse existing helpers/constants instead of re-hardcoding: _PRUNED_TOOL_PLACEHOLDER, _PRUNE_MIN_CHARS, _NEWEST_TURN_ONLY_BUDGET_KEYS, and _prune_stale_reasoning_replay (codex sidecar shrink, #71058 boundary). - Assistant-role messages without the summary metadata key are no longer truncatable by the summary-cap heuristic. - Caller passes budget so the estimator runs 3x, not 5x, per would-grow pass. --- agent/context_compressor.py | 69 ++++++++++++++++---- agent/conversation_compression.py | 4 +- tests/agent/test_salvage_grown_transcript.py | 66 ++++++++++++++++++- tests/run_agent/test_in_place_compaction.py | 4 +- 4 files changed, 127 insertions(+), 16 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 86baf5f1f3..4c48794fb2 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -350,10 +350,8 @@ _SUMMARY_END_MARKER = ( _MERGED_PRIOR_CONTEXT_HEADER = "[PRIOR CONTEXT — for reference only; not a new message]" _MERGED_SUMMARY_DELIMITER = "[END OF PRIOR CONTEXT — COMPACTION SUMMARY BELOW]" -_SALVAGE_TOOL_PLACEHOLDER = "[Old tool output cleared to save context space]" _SALVAGE_SUMMARY_MAX_CHARS = 8_000 _SALVAGE_KEEP_RECENT_TOOLS = 2 -_SALVAGE_REASONING_KEYS = ("reasoning", "reasoning_content", "reasoning_details") def _looks_like_compaction_summary(msg: Dict[str, Any], content: str) -> bool: @@ -363,11 +361,15 @@ def _looks_like_compaction_summary(msg: Dict[str, Any], content: str) -> bool: return False if content.startswith(_MERGED_PRIOR_CONTEXT_HEADER): return False - # Content heuristics alone must never authorize mutating a live user turn. - # Compressor-generated user-role summaries carry this private marker; - # ordinary user input does not. + # Content heuristics alone must never authorize mutating a live turn. + # Compressor-generated summaries carry this private marker; ordinary + # user input — and live assistant replies or kept tool bodies that + # merely quote a summary header/marker — do not. Tool messages are + # handled exclusively by the stub/keep-recent pass, never the cap. + if msg.get("role") == "tool": + return False if ( - msg.get("role") == "user" + msg.get("role") in ("user", "assistant") and not msg.get(COMPRESSED_SUMMARY_METADATA_KEY) ): return False @@ -380,9 +382,40 @@ def _looks_like_compaction_summary(msg: Dict[str, Any], content: str) -> bool: ) +def _salvage_reduce_todo_snapshot(out: List[Dict[str, Any]]) -> None: + """Last-resort shrink: reduce or drop the synthetic todo snapshot. + + The snapshot is the only in-transcript todo re-injection at a compaction + boundary, and since 7a16840add the pruned-skill reload notice is coupled + into the same string — so it is only touched when the cheaper shrink ops + could not get under budget. When the snapshot carries a reload notice, + keep just the notice (the coupling must survive salvage); otherwise drop + the row entirely. + """ + from agent.conversation_compression import _PRUNED_SKILL_RELOAD_NOTICE_HEADER + + for i in range(len(out) - 1, -1, -1): + msg = out[i] + if not isinstance(msg, dict): + continue + if msg.get("_todo_snapshot_synthetic") and msg.get("role") == "user": + content = msg.get("content") + notice_idx = ( + content.find(_PRUNED_SKILL_RELOAD_NOTICE_HEADER) + if isinstance(content, str) + else -1 + ) + if isinstance(content, str) and notice_idx >= 0: + msg["content"] = content[notice_idx:] + else: + del out[i] + return + + def salvage_grown_transcript( original: List[Dict[str, Any]], candidate: List[Dict[str, Any]], + budget: Optional[int] = None, ) -> Optional[List[Dict[str, Any]]]: """Mechanically shrink a compression candidate, or return ``None``. @@ -390,10 +423,17 @@ def salvage_grown_transcript( tool bodies, stale reasoning, or a synthetic todo snapshot tip the final candidate over the input size. Work on copies and admit the salvage only when the same rough estimator proves it is strictly smaller than the input. + + Shrink order is cheapest-information-loss first: stale reasoning keys and + codex replay sidecars, then old tool bodies, then an oversized summary cap. + The synthetic todo snapshot (which carries the pruned-skill reload notice, + see ``_salvage_reduce_todo_snapshot``) is only reduced as a LAST resort + when everything else still leaves the candidate at or over budget. """ if not candidate or not original: return None - budget = estimate_messages_tokens_rough(original) + if budget is None: + budget = estimate_messages_tokens_rough(original) if budget <= 0: return None @@ -404,8 +444,6 @@ def salvage_grown_transcript( if not isinstance(msg, dict): out.append(msg) continue - if msg.get("_todo_snapshot_synthetic") and msg.get("role") == "user": - continue copied = dict(msg) out.append(copied) role = copied.get("role") @@ -414,17 +452,18 @@ def salvage_grown_transcript( elif role == "assistant": last_assistant_idx = len(out) - 1 + salvage_reasoning_keys = _NEWEST_TURN_ONLY_BUDGET_KEYS + ("reasoning_details",) keep_tools = set(tool_indices[-_SALVAGE_KEEP_RECENT_TOOLS:]) for index, msg in enumerate(out): if not isinstance(msg, dict): continue if msg.get("role") == "assistant" and index != last_assistant_idx: - for key in _SALVAGE_REASONING_KEYS: + for key in salvage_reasoning_keys: msg.pop(key, None) if msg.get("role") == "tool" and index not in keep_tools: content = msg.get("content") - if isinstance(content, str) and len(content) > 200: - msg["content"] = _SALVAGE_TOOL_PLACEHOLDER + if isinstance(content, str) and len(content) > _PRUNE_MIN_CHARS: + msg["content"] = _PRUNED_TOOL_PLACEHOLDER content = msg.get("content") if ( isinstance(content, str) @@ -436,6 +475,12 @@ def salvage_grown_transcript( + "\n…[summary truncated so compaction can shrink]\n\n" + _SUMMARY_END_MARKER ) + # Heavier codex replay sidecars (encrypted reasoning blobs) — reuse the + # proven prune with its last-user-turn safety boundary (#71058). + _prune_stale_reasoning_replay(out) + + if estimate_messages_tokens_rough(out) >= budget: + _salvage_reduce_todo_snapshot(out) if not any( isinstance(message, dict) and message.get("role") == "user" diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 521c08d928..a2b2643bf3 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -3392,7 +3392,9 @@ def compress_context( # candidate over. Give it one mechanical salvage pass. from agent.context_compressor import salvage_grown_transcript - _salvaged = salvage_grown_transcript(messages, compressed) + _salvaged = salvage_grown_transcript( + messages, compressed, budget=_rough_in + ) if _salvaged is not None: _salv_est = estimate_messages_tokens_rough(_salvaged) if _salv_est < _rough_in: diff --git a/tests/agent/test_salvage_grown_transcript.py b/tests/agent/test_salvage_grown_transcript.py index 920dd85a1c..080aa08925 100644 --- a/tests/agent/test_salvage_grown_transcript.py +++ b/tests/agent/test_salvage_grown_transcript.py @@ -8,7 +8,12 @@ from agent.context_compressor import ( from agent.model_metadata import estimate_messages_tokens_rough -def test_salvage_stubs_old_tools_and_drops_todo(): +def test_salvage_stubs_old_tools_and_keeps_todo_when_stubbing_suffices(): + """Tool stubbing alone gets under budget → the todo snapshot survives. + + The snapshot is the only in-transcript todo re-injection at the boundary + (and may carry the pruned-skill reload notice), so it is last-resort only. + """ original = [ {"role": "user", "content": "go"}, {"role": "assistant", "content": "ok"}, @@ -29,12 +34,69 @@ def test_salvage_stubs_old_tools_and_drops_todo(): assert out is not None assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original) - assert not any(m.get("_todo_snapshot_synthetic") for m in out) + assert any(m.get("_todo_snapshot_synthetic") for m in out) tools = [m["content"] for m in out if m.get("role") == "tool"] assert tools[-1] == "keep-latest" assert any("cleared to save context space" in t for t in tools) +def test_salvage_drops_todo_only_as_last_resort(): + """When cheaper ops cannot get under budget, the snapshot is dropped.""" + original = [ + {"role": "user", "content": "please do the thing " + ("o" * 600)}, + {"role": "assistant", "content": "ok"}, + ] + grown = [ + {"role": "user", "content": "summary of the ask"}, + {"role": "assistant", "content": "ok"}, + { + "role": "user", + "content": "Current todos:\n- [ ] " + ("t" * 800), + "_todo_snapshot_synthetic": True, + }, + ] + assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original) + + out = salvage_grown_transcript(original, grown) + + assert out is not None + assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original) + assert not any(m.get("_todo_snapshot_synthetic") for m in out) + + +def test_salvage_last_resort_preserves_pruned_skill_reload_notice(): + """7a16840add couples the reload notice into the snapshot — it survives.""" + from agent.conversation_compression import _PRUNED_SKILL_RELOAD_NOTICE_HEADER + + notice = ( + f"{_PRUNED_SKILL_RELOAD_NOTICE_HEADER}\n" + "Reload with skill_view(name='example-skill') before acting." + ) + original = [ + {"role": "user", "content": "please do the thing " + ("o" * 3000)}, + {"role": "assistant", "content": "ok"}, + ] + grown = [ + {"role": "user", "content": "summary of the ask"}, + {"role": "assistant", "content": "ok"}, + { + "role": "user", + "content": "Current todos:\n- [ ] " + ("t" * 4000) + f"\n\n{notice}", + "_todo_snapshot_synthetic": True, + }, + ] + assert estimate_messages_tokens_rough(grown) > estimate_messages_tokens_rough(original) + + out = salvage_grown_transcript(original, grown) + + assert out is not None + assert estimate_messages_tokens_rough(out) < estimate_messages_tokens_rough(original) + snapshot_rows = [m for m in out if m.get("_todo_snapshot_synthetic")] + assert len(snapshot_rows) == 1 + assert snapshot_rows[0]["content"].startswith(_PRUNED_SKILL_RELOAD_NOTICE_HEADER) + assert "Current todos" not in snapshot_rows[0]["content"] + + def test_salvage_returns_none_when_nothing_can_shrink(): original = [{"role": "user", "content": "tiny"}] huge = [{"role": "user", "content": "X" * 200_000}] diff --git a/tests/run_agent/test_in_place_compaction.py b/tests/run_agent/test_in_place_compaction.py index e78e63ce52..93d8236151 100644 --- a/tests/run_agent/test_in_place_compaction.py +++ b/tests/run_agent/test_in_place_compaction.py @@ -345,7 +345,9 @@ class TestInPlaceAntiGrowthGuard: tool_bodies = [m.get("content") for m in compressed if m.get("role") == "tool"] assert any(isinstance(body, str) and body.startswith("keep-me") for body in tool_bodies) assert any("cleared to save context space" in (body or "") for body in tool_bodies) - assert not any(m.get("_todo_snapshot_synthetic") for m in compressed) + # Tool stubbing alone got under budget, so the todo snapshot (the + # only in-transcript todo re-injection) survives the salvage. + assert any(m.get("_todo_snapshot_synthetic") for m in compressed) def test_in_place_still_commits_shrinking_compression(self): """The guard must not block legitimate compressions — a result SMALLER From 7f2733b71c12009a6d67b74f829aa3dafd2db9bf Mon Sep 17 00:00:00 2001 From: Dhruv Modi Date: Wed, 19 Aug 2026 07:06:39 +0000 Subject: [PATCH 18/42] fix(model-metadata): converge output-cap retry on vLLM Fixes the retry loop that spins forever when a vLLM server rejects a request for having a max_tokens too big for what is left of the context window. The catch is that vLLM does not tell you how big your prompt actually is in that situation. It works the number backwards from the constraint it just failed, so you get: "requested 65536 output tokens and your prompt contains at least 36865 input tokens, for a total of at least 102401 tokens" That 36865 is just window + 1 - requested, and the total is always exactly window + 1. Subtracting it from the window hands back requested - 1 every single time, whatever the real prompt size is. parse_available_output_tokens_from_error believed it and returned requested - 1. conversation_loop then takes off its 64 token safety margin and retries, which walks the cap down 65 tokens at a time while the reported input walks up by the same 65: 65536 -> 65471 -> 65406 -> 65341 Three attempts is the default budget, so the session gives up with "Context length exceeded" having closed 195 tokens of a roughly 28000 token gap. Compression cannot save it either, because the input was never the problem, which is why the compressor keeps refusing with "summary would have GROWN". This is also what is behind the unexplained "input-token drift" in issue #61761. The input is not drifting. It is a derived number, and it moves because we moved max_tokens. So when that shape shows up (the "at least" wording, plus a budget that works out to exactly requested - 1), halve the requested cap instead. It is still guaranteed to sit under whatever was just rejected, and it converges on the first retry: 65536 -> 32768, which next to a real 36865 token prompt comes to 69633 against a 102400 window. Nothing else moves. A measured input is still trusted, and a genuine input overflow still returns None so the caller falls through to compression the way it always did. The existing test asserted the bogus 65535, so it is updated. Added tests for the measured input path, and for the retry actually converging. --- agent/model_metadata.py | 17 +++++++++++ tests/test_output_cap_parsing.py | 49 ++++++++++++++++++++++++++++++-- 2 files changed, 63 insertions(+), 3 deletions(-) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 57168fc960..b2e96d9f20 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -1726,11 +1726,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]: # Available output = window - input. When the input alone is at or over # the window this stays None, so the caller correctly falls through to # compression instead of futilely shrinking the output cap. + # + # Caveat: when max_tokens is the BINDING constraint, vLLM does not report + # the real prompt size at all. It back-computes a lower bound from the + # constraint itself -- "at least N input tokens" where + # N == window + 1 - requested_output -- so window - N is always exactly + # requested_output - 1. Subtracting the caller's safety margin then walks + # the cap down ~65 tokens per retry while the reported input walks up by + # the same amount, burning every compression attempt without ever fitting. + # Detect that degenerate case and halve the requested cap instead: it + # carries the same guarantee (strictly below what was rejected) and + # converges in one or two retries. _m_vllm_input = re.search( r'prompt contains (?:at least )?(\d+)\s*input tokens', error_lower ) if _m_ctx_tok and _m_vllm_input: _available = int(_m_ctx_tok.group(1)) - int(_m_vllm_input.group(1)) + _m_requested_out = re.search(r'requested (\d+)\s*output tokens', error_lower) + if 'at least' in error_lower and _m_requested_out: + _requested_out = int(_m_requested_out.group(1)) + if _available >= _requested_out - 1: + # The budget is derived from the constraint, not measured. + return max(1, _requested_out // 2) if _available >= 1: return _available diff --git a/tests/test_output_cap_parsing.py b/tests/test_output_cap_parsing.py index d477ef5da1..1825569976 100644 --- a/tests/test_output_cap_parsing.py +++ b/tests/test_output_cap_parsing.py @@ -121,10 +121,28 @@ class TestParseVllmTokenBasedOutputCap: "output tokens." ) - def test_vllm_token_based_format(self): - # available output = 131072 - 65537 = 65535 - assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 65535 + # Verbatim vLLM response where the input is MEASURED, not back-computed: + # window - input != requested - 1, so the reported figure is real. + _VLLM_MSG_REAL_INPUT = ( + "This model's maximum context length is 131072 tokens. However, you " + "requested 65536 output tokens and your prompt contains 100000 " + "input tokens, for a total of 165536 tokens. Please reduce the length " + "of the input prompt or the number of requested output tokens." + ) + def test_vllm_token_based_format(self): + # The reported input is a LOWER BOUND that vLLM back-computes from the + # constraint (65537 == 131072 + 1 - 65536), so window - input is just + # requested - 1 and carries no information about the real prompt. + # Halve the requested cap instead so the retry actually converges. + assert parse_available_output_tokens_from_error(self._VLLM_MSG) == 32768 + + def test_vllm_measured_input_is_trusted(self): + # When the input is measured rather than derived, use it as-is. + # available output = 131072 - 100000 = 31072 + assert parse_available_output_tokens_from_error( + self._VLLM_MSG_REAL_INPUT + ) == 31072 def test_vllm_retry_fits_inside_window(self): # The retried cap plus the reported input must fit in the window. @@ -132,3 +150,28 @@ class TestParseVllmTokenBasedOutputCap: assert available is not None assert available + 65537 <= 131072 + def test_vllm_retry_converges(self): + """The retry sequence must reach a working cap in a few attempts. + + Regression test for the 65-tokens-per-retry crawl: with a 102400 + window and a real prompt of ~37000 tokens, retrying from a 65536 cap + used to produce 65471 -> 65406 -> 65341 and exhaust the compression + budget without ever fitting. + """ + window, real_input, cap = 102400, 37000, 65536 + for _ in range(5): + if real_input + cap <= window: + break + # vLLM's message when max_tokens is the binding constraint. + msg = ( + f"This model's maximum context length is {window} tokens. " + f"However, you requested {cap} output tokens and your prompt " + f"contains at least {window + 1 - cap} input tokens, for a " + f"total of at least {window + 1} tokens." + ) + available = parse_available_output_tokens_from_error(msg) + assert available is not None + assert available < cap, "each retry must lower the cap" + cap = available + assert real_input + cap <= window, f"did not converge: cap={cap}" + From 99c980f466602d89b3b55150653136945e6f53c8 Mon Sep 17 00:00:00 2001 From: ekinnee Date: Thu, 20 Aug 2026 11:09:48 +0530 Subject: [PATCH 19/42] fix(model-metadata): parse 'exceeds model maximum output tokens' cap errors Recognizes the DeepSeek/OpenAI-compatible relay wording max_tokens (98304) exceeds model's maximum output tokens (65536) in both parse_available_output_tokens_from_error (returns the cap) and is_output_cap_error (keeps the 400 out of the compression death-loop). Salvaged from PR #72283; the conversation_loop early-clamp block was dropped in favor of routing through the existing output-cap handler (follow-up commit). --- agent/model_metadata.py | 20 ++++++++++++++++++++ tests/test_output_cap_parsing.py | 16 ++++++++++++++++ 2 files changed, 36 insertions(+) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index b2e96d9f20..0bbbaea061 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -1659,10 +1659,28 @@ def parse_available_output_tokens_from_error(error_msg: str) -> Optional[int]: # The input itself fits — this is purely an output-cap error, so reduce # max_tokens and retry; do NOT compress. "range of max_tokens should be" in error_lower + ) or ( + # OpenAI-compatible relays may reject a request whose output cap exceeds + # the model's separate completion-token limit, e.g. + # "max_tokens (98304) exceeds model's maximum output tokens (65536)" + # This is independent of the input context window. + "exceeds model" in error_lower + and "maximum output tokens" in error_lower ) if not is_output_cap_error: return None + # Generic model-output-cap form: + # "max_tokens (98304) exceeds model's maximum output tokens (65536)" + _m_max_output = re.search( + r'exceeds model(?:\'s)? maximum output tokens\s*\(?\s*(\d+)\s*\)?', + error_lower, + ) + if _m_max_output: + _cap = int(_m_max_output.group(1)) + if _cap >= 1: + return _cap + # DashScope / Alibaba range form: "Range of max_tokens should be [1, 65536]". # The upper bound is the available output cap. _m_range = re.search( @@ -1799,6 +1817,8 @@ def is_output_cap_error(error_msg: str) -> bool: or "should be" in error_lower # generic "max_tokens should be <= N" or "less than or equal" in error_lower or "must be" in error_lower + or ("exceeds model" in error_lower + and "maximum output tokens" in error_lower) ) if not output_cap_signal: return False diff --git a/tests/test_output_cap_parsing.py b/tests/test_output_cap_parsing.py index 1825569976..aa64d09a7e 100644 --- a/tests/test_output_cap_parsing.py +++ b/tests/test_output_cap_parsing.py @@ -68,6 +68,22 @@ class TestParseDashScopeOutputCap: assert parse_available_output_tokens_from_error(msg) == 32768 +class TestParseMaximumOutputTokensCap: + """Some OpenAI-compatible relays report the model's separate output cap.""" + + def test_parenthesized_max_output_cap(self): + msg = ( + "API call failed after 3 retries: [400]: max_tokens (98304) " + "exceeds model's maximum output tokens (65536)" + ) + assert parse_available_output_tokens_from_error(msg) == 65536 + + def test_parenthesized_max_output_cap_is_output_cap(self): + assert is_output_cap_error( + "max_tokens (98304) exceeds model's maximum output tokens (65536)" + ) is True + + class TestIsOutputCapError: """`is_output_cap_error` is the broader yes/no gate that keeps an output-cap 400 out of the compression death-loop even when we can't parse From b7e12decc6eefc0676f75c61b770af2b090341c6 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 20 Aug 2026 11:12:00 +0530 Subject: [PATCH 20/42] fix(agent): route relay-wrapped output-cap 429s into the output-cap handler Salvage follow-up for #72283: instead of a second pre-retry clamp block (which bypassed the #55546 clamp+compress path and broke its three regression tests), parse the output cap ONCE at classification time and: - exempt parseable wrapped output-cap 429s from the eager rate-limit provider fallback (a deterministic request-shape failure that failover cannot fix but the clamp fixes in one retry), and - widen is_context_length_error so they reach the SAME #55546 clamp+compress recovery as plain output-cap 400s. Adds both #72283 regression scenarios plus an ordering guard proving a NON-EMPTY fallback chain does not consume the wrapped 429 (fallback slot unspent, model unchanged). 119 fallback/rate-limit tests green. --- agent/conversation_loop.py | 22 +++++- tests/run_agent/test_run_agent.py | 112 ++++++++++++++++++++++++++++++ 2 files changed, 133 insertions(+), 1 deletion(-) diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index e0250124ad..768d8e019e 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -5304,6 +5304,21 @@ def run_conversation( FailoverReason.billing, FailoverReason.upstream_rate_limit, } + # Relay-wrapped output-cap errors: some gateways wrap an + # upstream "[400]: max_tokens (...) exceeds model's maximum + # output tokens (...)" as HTTP 429, which classifies as + # rate_limit. The failure is a deterministic request-shape + # problem — falling back to another provider (or burning + # generic retries) can't fix it, but the output-cap clamp + # below can, in one retry (#72281). Parse once here; the + # result gates both the eager-fallback exemption and the + # widened is_context_length_error entry, and is reused as + # available_out inside the handler. + _wrapped_output_cap_budget = ( + parse_available_output_tokens_from_error(error_msg) + if classified.reason == FailoverReason.rate_limit + else None + ) _is_transport_failure = classified.reason in { FailoverReason.timeout, FailoverReason.overloaded, @@ -5321,7 +5336,7 @@ def run_conversation( if _is_zai_coding_overload: max_retries = max(max_retries, zai_coding_overload_retry_ceiling()) _should_fallback = ( - is_rate_limited + (is_rate_limited and _wrapped_output_cap_budget is None) or (_is_transport_failure and retry_count >= 2) ) if _should_fallback and agent._fallback_index < len(agent._fallback_chain): @@ -5614,6 +5629,11 @@ def run_conversation( # server disconnect + large session pattern (#2153). is_context_length_error = ( classified.reason == FailoverReason.context_overflow + # Relay-wrapped output-cap 429s (parsed once above, where + # the eager-fallback exemption is gated) route into the + # output-cap clamp below instead of provider failover or + # generic retries (#72281). + or _wrapped_output_cap_budget is not None ) if is_context_length_error: diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py index 68eb70bcc8..a72da553be 100644 --- a/tests/run_agent/test_run_agent.py +++ b/tests/run_agent/test_run_agent.py @@ -4535,6 +4535,118 @@ class TestRunConversation: assert agent.context_compressor.context_length == 200_000 mock_compress.assert_called_once() + def test_output_cap_retry_before_generic_retry_exhaustion(self, agent): + """Provider max-output-cap 400s clamp via the output-cap handler, not + the generic retry loop ("failed after 3 retries"). + """ + self._setup_agent(agent) + agent.api_mode = "chat_completions" + agent.provider = "deepseek" + agent.base_url = "https://api.deepseek.com/v1" + agent.model = "deepseek-v4-flash" + agent.max_tokens = 98_304 + agent.compression_enabled = True + agent.context_compressor.context_length = 200_000 + agent.context_compressor.should_compress = MagicMock(return_value=False) + + error_msg = ( + "[400]: max_tokens (98304) exceeds model's maximum output tokens " + "(65536) for model deepseek-v4-flash " + "(ref: 7735422e-9cb4-4075-a779-dfecb3204a0e)" + ) + exc = Exception(error_msg) + exc.status_code = 400 + exc.code = 400 + + ok_resp = _mock_response(content="done", finish_reason="stop") + agent.client.chat.completions.create.side_effect = [exc, ok_resp] + + mock_compress = MagicMock(return_value=( + [{"role": "user", "content": "hello"}], + "You are helpful.", + )) + with ( + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + patch.object(agent.context_compressor, "update_model"), + patch.object(agent, "_compress_context", mock_compress), + ): + result = agent.run_conversation("hello") + + assert len(agent.client.chat.completions.create.call_args_list) == 2 + second_call = agent.client.chat.completions.create.call_args_list[1].kwargs + assert result["completed"] is True + assert second_call["max_tokens"] <= 65_472 + assert agent.context_compressor.context_length == 200_000 + + def test_output_cap_retry_when_gateway_wraps_error_as_rate_limit(self, agent): + """Some relays wrap the upstream max-output 400 as HTTP 429. The + parseable output cap must still route into the output-cap handler + instead of burning generic rate-limit retries (#72281). + """ + self._run_wrapped_429_output_cap(agent, fallback_chain=[]) + + def test_wrapped_output_cap_429_not_consumed_by_eager_fallback(self, agent): + """With a NON-EMPTY fallback chain, the eager rate-limit fallback must + NOT consume the wrapped output-cap 429 — the failure is a deterministic + request-shape problem the clamp fixes in one retry; switching provider + burns a fallback slot for nothing (#72281 ordering guard). + """ + self._run_wrapped_429_output_cap( + agent, + fallback_chain=[{"provider": "openrouter", "model": "anthropic/claude-sonnet-4"}], + ) + + def _run_wrapped_429_output_cap(self, agent, *, fallback_chain): + self._setup_agent(agent) + agent.api_mode = "chat_completions" + agent.provider = "custom" + agent.base_url = "http://192.168.1.254:20128/v1" + agent.model = "deepseekv4flash" + agent.max_tokens = 98_304 + agent.compression_enabled = True + agent._fallback_chain = fallback_chain + agent._fallback_index = 0 + agent.context_compressor.context_length = 200_000 + agent.context_compressor.should_compress = MagicMock(return_value=False) + + error_msg = ( + "Error code: 429 - {'error': {'message': \"[400]: max_tokens " + "(98304) exceeds model's maximum output tokens (65536) for model " + "deepseek-v4-flash (ref: 37bde60f-44e7-44e2-b995-4af17fba6d6b)\", " + "'type': 'rate_limit_error', 'code': 'rate_limit_exceeded'}}" + ) + exc = Exception(error_msg) + exc.status_code = 429 + exc.code = "rate_limit_exceeded" + + ok_resp = _mock_response(content="done", finish_reason="stop") + agent.client.chat.completions.create.side_effect = [exc, ok_resp] + + mock_compress = MagicMock(return_value=( + [{"role": "user", "content": "hello"}], + "You are helpful.", + )) + with ( + patch.object(agent, "_persist_session"), + patch.object(agent, "_save_trajectory"), + patch.object(agent, "_cleanup_task_resources"), + patch.object(agent.context_compressor, "update_model"), + patch.object(agent, "_compress_context", mock_compress), + ): + result = agent.run_conversation("hello") + + assert len(agent.client.chat.completions.create.call_args_list) == 2 + second_call = agent.client.chat.completions.create.call_args_list[1].kwargs + assert result["completed"] is True + assert second_call["max_tokens"] <= 65_472 + assert agent.context_compressor.context_length == 200_000 + # The clamp, not provider failover, must have recovered: no fallback + # slot consumed and the model unchanged. + assert agent._fallback_index == 0 + assert agent.model == "deepseekv4flash" + def test_output_cap_retry_with_large_api_only_content(self, agent): """When a large system prompt makes api_messages huge while persisted messages stay tiny, the retry cap must still respect provider From 67029492d472dea515b56084a258270ab093ae64 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 20 Aug 2026 11:28:47 +0530 Subject: [PATCH 21/42] chore: map salvaged contributor emails (Dhruv7201, ekinnee) --- contributors/emails/dhruv.modi2345@gmail.com | 2 ++ contributors/emails/erick@kinnee.net | 2 ++ 2 files changed, 4 insertions(+) create mode 100644 contributors/emails/dhruv.modi2345@gmail.com create mode 100644 contributors/emails/erick@kinnee.net diff --git a/contributors/emails/dhruv.modi2345@gmail.com b/contributors/emails/dhruv.modi2345@gmail.com new file mode 100644 index 0000000000..543909fb6c --- /dev/null +++ b/contributors/emails/dhruv.modi2345@gmail.com @@ -0,0 +1,2 @@ +Dhruv7201 +# PR #89923 salvage (vLLM output-cap convergence) diff --git a/contributors/emails/erick@kinnee.net b/contributors/emails/erick@kinnee.net new file mode 100644 index 0000000000..b6ef2f887a --- /dev/null +++ b/contributors/emails/erick@kinnee.net @@ -0,0 +1,2 @@ +ekinnee +# PR #72283 salvage (maximum output tokens parsing) From 1a8fea3ce24dcec339c4532efd9c906b72bf6c8d Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Mon, 17 Aug 2026 18:15:01 +0530 Subject: [PATCH 22/42] fix(cli): skip Kitty keyboard protocol push for Ghostty, use modifyOtherKeys only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ghostty's Kitty disambiguate-mode implementation strips the Alt modifier from the Backspace key — Option+Backspace arrives as bare \x7f instead of the expected \x1b[27;3;127~, breaking backward-kill-word. This was a regression introduced when PR #87630 re-added the CSI >1u Kitty protocol push for all allowlisted terminals including Ghostty. Under modifyOtherKeys mode (CSI >4;2m), Ghostty correctly sends \x1b[27;3;127~ for Option+Backspace, which the alias table in pt_input_extras already maps to (Escape, ControlH) = backward-kill-word. Fix: for Ghostty only, push just modifyOtherKeys and skip the Kitty protocol push. All other terminals (iTerm2, WezTerm, kitty, tmux, VS Code) still get the full dual-protocol push. Ghostty upstream tracking: discussion #9560, issue #9895 (cmd+backspace variant of the same root cause). --- cli.py | 28 ++++++++++++- tests/cli/test_ctrl_enter_newline.py | 60 ++++++++++++++++++++++++++++ 2 files changed, 86 insertions(+), 2 deletions(-) diff --git a/cli.py b/cli.py index 97d5719fd4..b5cd92059b 100644 --- a/cli.py +++ b/cli.py @@ -4095,6 +4095,13 @@ _TERMINAL_INPUT_MODE_RESET_SEQ = ( "\x1b[?25h" # ensure cursor visible ) _EXTENDED_ENTER_KEYS_SEQ = "\x1b[>1u\x1b[>4;2m" +# Ghostty: push ONLY modifyOtherKeys, not the Kitty keyboard protocol. +# Ghostty's Kitty disambiguate-mode implementation strips the Alt modifier +# from the Backspace key — Option+Backspace arrives as bare \x7f instead of +# the expected \x1b[27;3;127~, breaking backward-kill-word (#87630 +# regression). modifyOtherKeys mode works correctly on Ghostty and covers +# every modified key combo the alias table handles. +_GHOSTTY_EXTENDED_ENTER_KEYS_SEQ = "\x1b[>4;2m" _BACKSLASH_LINE_CONTINUATION_RE = re.compile(r"\\[ \t]*$") @@ -4148,20 +4155,37 @@ def _enable_extended_enter_keys(output=None, env: Optional[Mapping[str, str]] = Ctrl+C, which is handled by prompt_toolkit's ``c-c`` binding (raw mode clears ISIG, so the kernel INTR path was never in play for the CLI). + Ghostty exception: Ghostty's Kitty disambiguate mode strips the Alt + modifier from the Backspace key, so Option+Backspace arrives as bare + \x7f instead of \x1b[27;3;127~, breaking backward-kill-word. Ghostty + implements modifyOtherKeys correctly, so for Ghostty we push only + modifyOtherKeys and skip the Kitty protocol push. + The exit reset sequence pops/resets both modes, so this is safe across normal exits, Ctrl+C, and SIGTERM cleanup. """ if not _terminal_supports_extended_enter_keys(env): return False try: + # Ghostty: skip Kitty protocol, use only modifyOtherKeys. + if env is None: + env = os.environ + term_program = (env.get("TERM_PROGRAM") or "").strip() + term = (env.get("TERM") or "").strip().lower() + is_ghostty = ( + term_program == "ghostty" + or term == "xterm-ghostty" + ) + seq = _GHOSTTY_EXTENDED_ENTER_KEYS_SEQ if is_ghostty else _EXTENDED_ENTER_KEYS_SEQ + target = output if target is not None and hasattr(target, "write_raw"): - target.write_raw(_EXTENDED_ENTER_KEYS_SEQ) + target.write_raw(seq) target.flush() return True stream = sys.stdout if stream is not None and stream.isatty(): - stream.write(_EXTENDED_ENTER_KEYS_SEQ) + stream.write(seq) stream.flush() return True except Exception: diff --git a/tests/cli/test_ctrl_enter_newline.py b/tests/cli/test_ctrl_enter_newline.py index ba7cb2f564..917a2b393d 100644 --- a/tests/cli/test_ctrl_enter_newline.py +++ b/tests/cli/test_ctrl_enter_newline.py @@ -153,6 +153,66 @@ def test_unknown_terminal_does_not_enable_extended_enter_keys(): assert cli_mod._terminal_supports_extended_enter_keys({"TERM_PROGRAM": "unknown"}) is False +# --------------------------------------------------------------------------- +# Ghostty: must push ONLY modifyOtherKeys, not the Kitty keyboard protocol. +# Ghostty's Kitty disambiguate mode strips the Alt modifier from Backspace, +# so Option+Backspace arrives as bare \x7f instead of \x1b[27;3;127~, +# breaking backward-kill-word. modifyOtherKeys mode works correctly. +# --------------------------------------------------------------------------- +class _FakeOutput: + """Minimal output object with write_raw + flush for _enable_extended_enter_keys.""" + def __init__(self): + self.written = b"" + def write_raw(self, data): + self.written += data.encode() if isinstance(data, str) else data + def flush(self): + pass + + +def test_ghostty_uses_modify_other_keys_only(): + """Ghostty must NOT push the Kitty keyboard protocol (CSI >1u).""" + import cli as cli_mod + + out = _FakeOutput() + result = cli_mod._enable_extended_enter_keys( + output=out, + env={"TERM_PROGRAM": "ghostty", "TERM": "xterm-ghostty"}, + ) + assert result is True + # Must contain modifyOtherKeys push ... + assert b"\x1b[>4;2m" in out.written + # ... but NOT the Kitty protocol push. + assert b"\x1b[>1u" not in out.written + + +def test_ghostty_via_term_var_uses_modify_other_keys_only(): + """xterm-ghostty TERM (without TERM_PROGRAM) also skips Kitty protocol.""" + import cli as cli_mod + + out = _FakeOutput() + result = cli_mod._enable_extended_enter_keys( + output=out, + env={"TERM": "xterm-ghostty"}, + ) + assert result is True + assert b"\x1b[>4;2m" in out.written + assert b"\x1b[>1u" not in out.written + + +def test_non_ghostty_terminals_still_push_kitty_protocol(): + """iTerm2 and others still get the full dual-protocol push.""" + import cli as cli_mod + + out = _FakeOutput() + result = cli_mod._enable_extended_enter_keys( + output=out, + env={"TERM_PROGRAM": "iTerm.app", "TERM": "xterm-256color"}, + ) + assert result is True + assert b"\x1b[>1u" in out.written + assert b"\x1b[>4;2m" in out.written + + @pytest.mark.linux_only def test_proc_version_microsoft_marker_preserves_newline(): """WSL detection via /proc when env vars are scrubbed (sudo etc.). From 45f11263bd35140dcc749ae82246538d9dea354d Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 20 Aug 2026 11:24:01 +0530 Subject: [PATCH 23/42] fix(tui): skip the kitty protocol push for Ghostty in the Ink TUI too Widen the cli.py Ghostty exception to the sibling sites the review found: the Ink TUI pushes CSI >1u at raw-mode entry (App.tsx), on alt-screen exit, and on the extended-keys re-assert path (ink.tsx) for every EXTENDED_KEYS_TERMINALS entry including ghostty - same Alt-stripping bug. New skipKittyKeyboardProtocol() helper in terminal.ts gates the ENABLE push at all 3 sites; the DISABLE (pop) stays unconditional since popping an empty stack is a spec no-op. Also fix the cli.py comment citing the modifyOtherKeys encoding where the kitty CSI-u form (ESC[127;3u) is what the broken path expected, dedupe the quadruplicated Ghostty comment, and update the stale 'mirroring the Ink TUI' docstring. 7 new vitest cases. --- cli.py | 57 +++++++++++-------- tests/cli/test_ctrl_enter_newline.py | 17 ++++-- .../hermes-ink/src/ink/components/App.tsx | 9 ++- ui-tui/packages/hermes-ink/src/ink/ink.tsx | 13 ++++- .../hermes-ink/src/ink/terminal.test.ts | 33 +++++++++++ .../packages/hermes-ink/src/ink/terminal.ts | 10 ++++ 6 files changed, 106 insertions(+), 33 deletions(-) diff --git a/cli.py b/cli.py index b5cd92059b..329f6efa7e 100644 --- a/cli.py +++ b/cli.py @@ -4094,19 +4094,36 @@ _TERMINAL_INPUT_MODE_RESET_SEQ = ( "\x1b[0m" # reset text attributes "\x1b[?25h" # ensure cursor visible ) -_EXTENDED_ENTER_KEYS_SEQ = "\x1b[>1u\x1b[>4;2m" -# Ghostty: push ONLY modifyOtherKeys, not the Kitty keyboard protocol. -# Ghostty's Kitty disambiguate-mode implementation strips the Alt modifier -# from the Backspace key — Option+Backspace arrives as bare \x7f instead of -# the expected \x1b[27;3;127~, breaking backward-kill-word (#87630 -# regression). modifyOtherKeys mode works correctly on Ghostty and covers -# every modified key combo the alias table handles. -_GHOSTTY_EXTENDED_ENTER_KEYS_SEQ = "\x1b[>4;2m" +_KITTY_KEYBOARD_PUSH_SEQ = "\x1b[>1u" +_MODIFY_OTHER_KEYS_SEQ = "\x1b[>4;2m" +_EXTENDED_ENTER_KEYS_SEQ = _KITTY_KEYBOARD_PUSH_SEQ + _MODIFY_OTHER_KEYS_SEQ _BACKSLASH_LINE_CONTINUATION_RE = re.compile(r"\\[ \t]*$") +def _is_ghostty_terminal(env: Optional[Mapping[str, str]] = None) -> bool: + """Whether the terminal is Ghostty (either detection path). + + Ghostty must be pushed ONLY modifyOtherKeys, not the Kitty keyboard + protocol: its Kitty disambiguate-mode implementation strips the Alt + modifier from the Backspace key, so Option+Backspace arrives as bare + \\x7f instead of the CSI-u form ``\\x1b[127;3u`` the protocol calls for + (upstream Ghostty bug), breaking backward-kill-word (#87630 + regression). Ghostty implements modifyOtherKeys correctly (it then + emits ``\\x1b[27;3;127~``, which the alias table also maps). + + Matches exactly the two conditions that admit Ghostty through + ``_terminal_supports_extended_enter_keys``. + """ + if env is None: + env = os.environ + return ( + (env.get("TERM_PROGRAM") or "").strip() == "ghostty" + or (env.get("TERM") or "").strip().lower() == "xterm-ghostty" + ) + + def _terminal_supports_extended_enter_keys(env: Optional[Mapping[str, str]] = None) -> bool: """Whether it is safe/useful to request modified Enter key reporting. @@ -4138,7 +4155,9 @@ def _enable_extended_enter_keys(output=None, env: Optional[Mapping[str, str]] = Writes the Kitty keyboard protocol push (CSI >1u, disambiguate mode) AND xterm modifyOtherKeys level 2 (CSI >4;2m), mirroring the Ink TUI — - terminals honor whichever protocol they implement. Both are needed: + terminals honor whichever protocol they implement (except Ghostty, which + gets only modifyOtherKeys; see the Ghostty exception below). Both are + needed: kitty-the-terminal removed modifyOtherKeys support entirely (it only speaks its own protocol), while tmux/VS Code only accept modifyOtherKeys. @@ -4155,29 +4174,17 @@ def _enable_extended_enter_keys(output=None, env: Optional[Mapping[str, str]] = Ctrl+C, which is handled by prompt_toolkit's ``c-c`` binding (raw mode clears ISIG, so the kernel INTR path was never in play for the CLI). - Ghostty exception: Ghostty's Kitty disambiguate mode strips the Alt - modifier from the Backspace key, so Option+Backspace arrives as bare - \x7f instead of \x1b[27;3;127~, breaking backward-kill-word. Ghostty - implements modifyOtherKeys correctly, so for Ghostty we push only - modifyOtherKeys and skip the Kitty protocol push. + Ghostty exception: pushes only modifyOtherKeys — see + ``_is_ghostty_terminal`` for the full rationale (#87630). The exit reset sequence pops/resets both modes, so this is safe across normal exits, Ctrl+C, and SIGTERM cleanup. """ if not _terminal_supports_extended_enter_keys(env): return False + # Ghostty exception: only modifyOtherKeys — see _is_ghostty_terminal. + seq = _MODIFY_OTHER_KEYS_SEQ if _is_ghostty_terminal(env) else _EXTENDED_ENTER_KEYS_SEQ try: - # Ghostty: skip Kitty protocol, use only modifyOtherKeys. - if env is None: - env = os.environ - term_program = (env.get("TERM_PROGRAM") or "").strip() - term = (env.get("TERM") or "").strip().lower() - is_ghostty = ( - term_program == "ghostty" - or term == "xterm-ghostty" - ) - seq = _GHOSTTY_EXTENDED_ENTER_KEYS_SEQ if is_ghostty else _EXTENDED_ENTER_KEYS_SEQ - target = output if target is not None and hasattr(target, "write_raw"): target.write_raw(seq) diff --git a/tests/cli/test_ctrl_enter_newline.py b/tests/cli/test_ctrl_enter_newline.py index 917a2b393d..a7eaef7726 100644 --- a/tests/cli/test_ctrl_enter_newline.py +++ b/tests/cli/test_ctrl_enter_newline.py @@ -154,10 +154,8 @@ def test_unknown_terminal_does_not_enable_extended_enter_keys(): # --------------------------------------------------------------------------- -# Ghostty: must push ONLY modifyOtherKeys, not the Kitty keyboard protocol. -# Ghostty's Kitty disambiguate mode strips the Alt modifier from Backspace, -# so Option+Backspace arrives as bare \x7f instead of \x1b[27;3;127~, -# breaking backward-kill-word. modifyOtherKeys mode works correctly. +# Ghostty: must push ONLY modifyOtherKeys, not the Kitty keyboard protocol — +# see cli._is_ghostty_terminal for the full rationale (#87630). # --------------------------------------------------------------------------- class _FakeOutput: """Minimal output object with write_raw + flush for _enable_extended_enter_keys.""" @@ -237,3 +235,14 @@ def test_proc_version_microsoft_marker_preserves_newline(): # --------------------------------------------------------------------------- # install_ctrl_enter_alias() — ANSI sequence mappings for enhanced terminals # --------------------------------------------------------------------------- + + +def test_is_ghostty_terminal_detection_paths(): + """_is_ghostty_terminal matches exactly the two allowlist conditions.""" + import cli as cli_mod + + assert cli_mod._is_ghostty_terminal({"TERM_PROGRAM": "ghostty"}) is True + assert cli_mod._is_ghostty_terminal({"TERM": "xterm-ghostty"}) is True + assert cli_mod._is_ghostty_terminal({"TERM": "XTERM-GHOSTTY"}) is True + assert cli_mod._is_ghostty_terminal({"TERM_PROGRAM": "iTerm.app"}) is False + assert cli_mod._is_ghostty_terminal({}) is False diff --git a/ui-tui/packages/hermes-ink/src/ink/components/App.tsx b/ui-tui/packages/hermes-ink/src/ink/components/App.tsx index 10b3482f4c..b432e525e6 100644 --- a/ui-tui/packages/hermes-ink/src/ink/components/App.tsx +++ b/ui-tui/packages/hermes-ink/src/ink/components/App.tsx @@ -28,6 +28,7 @@ import { setTerminalBackgroundHex, setTerminalForegroundHex, setXtversionName, + skipKittyKeyboardProtocol, supportsExtendedKeys } from '../terminal.js' import { @@ -333,9 +334,13 @@ export default class App extends PureComponent { // distinguishable from ctrl+. We write both the kitty stack // push (CSI >1u) and xterm modifyOtherKeys level 2 (CSI >4;2m) — // terminals honor whichever they implement (tmux only accepts the - // latter). + // latter). Ghostty gets only modifyOtherKeys — its kitty + // disambiguate mode strips Alt from Backspace (see + // skipKittyKeyboardProtocol). if (supportsExtendedKeys()) { - this.props.stdout.write(ENABLE_KITTY_KEYBOARD) + if (!skipKittyKeyboardProtocol()) { + this.props.stdout.write(ENABLE_KITTY_KEYBOARD) + } this.props.stdout.write(ENABLE_MODIFY_OTHER_KEYS) } diff --git a/ui-tui/packages/hermes-ink/src/ink/ink.tsx b/ui-tui/packages/hermes-ink/src/ink/ink.tsx index c7ab0530eb..0254a81e69 100644 --- a/ui-tui/packages/hermes-ink/src/ink/ink.tsx +++ b/ui-tui/packages/hermes-ink/src/ink/ink.tsx @@ -77,6 +77,7 @@ import { } from './selection.js' import { needsAltScreenResizeScrollbackClear, + skipKittyKeyboardProtocol, supportsExtendedKeys, SYNC_OUTPUT_SUPPORTED, type Terminal, @@ -733,7 +734,11 @@ export default class Ink { // without the pop we'd accumulate depth on each editor round-trip). this.options.stdout.write( '\x1b[?1004h' + - (supportsExtendedKeys() ? DISABLE_KITTY_KEYBOARD + ENABLE_KITTY_KEYBOARD + ENABLE_MODIFY_OTHER_KEYS : '') + (supportsExtendedKeys() + ? DISABLE_KITTY_KEYBOARD + + (skipKittyKeyboardProtocol() ? '' : ENABLE_KITTY_KEYBOARD) + + ENABLE_MODIFY_OTHER_KEYS + : '') ) } onRender() { @@ -1472,7 +1477,11 @@ export default class Ink { // Pop-before-push keeps Kitty stack depth at 1 instead of accumulating // on each call. if (supportsExtendedKeys()) { - this.options.stdout.write(DISABLE_KITTY_KEYBOARD + ENABLE_KITTY_KEYBOARD + ENABLE_MODIFY_OTHER_KEYS) + this.options.stdout.write( + DISABLE_KITTY_KEYBOARD + + (skipKittyKeyboardProtocol() ? '' : ENABLE_KITTY_KEYBOARD) + + ENABLE_MODIFY_OTHER_KEYS + ) } if (!this.altScreenActive) { diff --git a/ui-tui/packages/hermes-ink/src/ink/terminal.test.ts b/ui-tui/packages/hermes-ink/src/ink/terminal.test.ts index 800f427840..d41902e486 100644 --- a/ui-tui/packages/hermes-ink/src/ink/terminal.test.ts +++ b/ui-tui/packages/hermes-ink/src/ink/terminal.test.ts @@ -74,3 +74,36 @@ describe('writeDiffToTerminal sync-marker gating (#66490 main-screen gap)', () = expect(out).toContain('frame') }) }) + +describe('skipKittyKeyboardProtocol', () => { + // Ghostty's kitty disambiguate mode strips the Alt modifier from + // Backspace (Option+Backspace arrives as bare \x7f instead of CSI-u + // \x1b[127;3u), so Ghostty must get only the modifyOtherKeys push. + // Mirrors cli.py's _GHOSTTY_EXTENDED_ENTER_KEYS_SEQ exception. + it('skips the kitty protocol push for ghostty', async () => { + const { env } = await import('../utils/env.js') + const { skipKittyKeyboardProtocol } = await import('./terminal.js') + const saved = env.terminal + try { + env.terminal = 'ghostty' + expect(skipKittyKeyboardProtocol()).toBe(true) + } finally { + env.terminal = saved + } + }) + + it.each(['iTerm.app', 'kitty', 'WezTerm', 'tmux', 'windows-terminal', 'vscode'])( + 'keeps the dual push for %s', + async terminal => { + const { env } = await import('../utils/env.js') + const { skipKittyKeyboardProtocol } = await import('./terminal.js') + const saved = env.terminal + try { + env.terminal = terminal + expect(skipKittyKeyboardProtocol()).toBe(false) + } finally { + env.terminal = saved + } + } + ) +}) diff --git a/ui-tui/packages/hermes-ink/src/ink/terminal.ts b/ui-tui/packages/hermes-ink/src/ink/terminal.ts index 1f709c4e5b..81df8be640 100644 --- a/ui-tui/packages/hermes-ink/src/ink/terminal.ts +++ b/ui-tui/packages/hermes-ink/src/ink/terminal.ts @@ -326,6 +326,16 @@ export function supportsExtendedKeys(): boolean { return EXTENDED_KEYS_TERMINALS.includes(env.terminal ?? '') } +/** True when the Kitty keyboard protocol push (CSI >1u) must be skipped for + * this terminal even though extended keys are supported. Ghostty's Kitty + * disambiguate-mode implementation strips the Alt modifier from Backspace — + * Option+Backspace arrives as bare \x7f instead of CSI-u \x1b[127;3u — + * breaking backward-kill-word. Ghostty implements modifyOtherKeys correctly, + * so it gets only that push (mirrors cli.py's Ghostty exception). */ +export function skipKittyKeyboardProtocol(): boolean { + return env.terminal === 'ghostty' +} + /** True if the terminal scrolls the viewport when it receives cursor-up * sequences that reach above the visible area. On Windows, conhost's * SetConsoleCursorPosition follows the cursor into scrollback From b8850e17b5ff92cca0f77e6669fafff3e203a02b Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 20 Aug 2026 01:04:12 -0500 Subject: [PATCH 24/42] fix(desktop): floating sidebar overlays keep an opaque background under glass --- .../components/pane-shell/tree/renderer/narrow-overlays.tsx | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/narrow-overlays.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/narrow-overlays.tsx index 02ef04adaa..3f11c0e797 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/narrow-overlays.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/narrow-overlays.tsx @@ -150,6 +150,10 @@ export function NarrowOverlays() { ? 'left-0 border-r border-(--ui-stroke-secondary)' : 'right-0 border-l border-(--ui-stroke-secondary)' )} + // Floats OVER the layout, so under glass its surface must mask the + // panes beneath it — a see-through overlay reads as text bleeding + // through text. Contract: `[data-glass-opaque]` in styles.css. + data-glass-opaque="" onMouseLeave={() => setReveal(current => (current?.pinned ? current : null))} // Match the pane's docked width (sessions ~237px, files its rail // width) instead of a fat fixed 20rem — capped for tiny screens. From b38c40319d0598f0ca67f19f6e5302cf306960c3 Mon Sep 17 00:00:00 2001 From: Cassie Gray Date: Sat, 15 Aug 2026 22:07:31 -0400 Subject: [PATCH 25/42] fix(cli): align hermes memory status and docs with memory tool gate --- hermes_cli/memory_setup.py | 5 ++-- tests/hermes_cli/test_memory_setup.py | 32 ++++++++++++++++++++++ website/docs/user-guide/features/memory.md | 2 +- 3 files changed, 36 insertions(+), 3 deletions(-) diff --git a/hermes_cli/memory_setup.py b/hermes_cli/memory_setup.py index 0087d61fd5..5113b4937c 100644 --- a/hermes_cli/memory_setup.py +++ b/hermes_cli/memory_setup.py @@ -490,10 +490,11 @@ def cmd_status(args) -> None: user_mark = "enabled ✓" if user_profile_enabled else "disabled ✗" # Check if the memory tool is enabled for the CLI platform via the - # canonical resolver (handles composite toolsets like hermes-cli). + # canonical resolver and respects the check_fn gate when both stores are disabled. from hermes_cli.tools_config import _get_platform_tools + from tools.memory_tool import check_memory_requirements cli_tools = _get_platform_tools(config, "cli", include_default_mcp_servers=False) - memory_tool_enabled = "memory" in cli_tools + memory_tool_enabled = ("memory" in cli_tools) and check_memory_requirements() tool_mark = "enabled ✓" if memory_tool_enabled else "disabled ✗" print("\nMemory status\n" + "─" * 40) diff --git a/tests/hermes_cli/test_memory_setup.py b/tests/hermes_cli/test_memory_setup.py index 8bcb6a3b16..674a9b51d1 100644 --- a/tests/hermes_cli/test_memory_setup.py +++ b/tests/hermes_cli/test_memory_setup.py @@ -96,3 +96,35 @@ def test_install_dependencies_force_reinstalls_versioned_specs(tmp_path, monkeyp assert installed, "force=True must reach the install step" assert any("mem0ai>=2.0.10,<3" in specs for specs in installed) + + +def test_cmd_status_memory_tool_gate_disabled(capsys, monkeypatch): + """When both memory stores are disabled, Memory status reports memory tool as disabled.""" + monkeypatch.setattr( + "hermes_cli.config.load_config", + lambda: {"memory": {"memory_enabled": False, "user_profile_enabled": False}}, + ) + monkeypatch.setattr(memory_setup, "_get_available_providers", lambda: []) + + memory_setup.cmd_status(SimpleNamespace()) + + captured = capsys.readouterr().out + assert "Memory tool: disabled ✗" in captured + assert "Memory injection: disabled ✗" in captured + assert "User profile: disabled ✗" in captured + + +def test_cmd_status_memory_tool_gate_enabled(capsys, monkeypatch): + """When at least one memory store is enabled, Memory status reports memory tool as enabled.""" + monkeypatch.setattr( + "hermes_cli.config.load_config", + lambda: {"memory": {"memory_enabled": True, "user_profile_enabled": False}}, + ) + monkeypatch.setattr(memory_setup, "_get_available_providers", lambda: []) + + memory_setup.cmd_status(SimpleNamespace()) + + captured = capsys.readouterr().out + assert "Memory tool: enabled ✓" in captured + assert "Memory injection: enabled ✓" in captured + assert "User profile: disabled ✗" in captured diff --git a/website/docs/user-guide/features/memory.md b/website/docs/user-guide/features/memory.md index cf575508ab..11a1f2376f 100644 --- a/website/docs/user-guide/features/memory.md +++ b/website/docs/user-guide/features/memory.md @@ -266,7 +266,7 @@ first, set `memory.write_approval: true`. It's a simple on/off gate applied to | `false` (default) | Write freely — the gate is off (the pre-gate behaviour). | | `true` | Require approval before anything is saved. In the interactive CLI, foreground writes prompt you inline (entries are small enough to read in full). Everywhere else — messaging platforms, scripts, and the background self-improvement review — writes are **staged** for review with `/memory pending`. | -> To turn memory off entirely (not just gate it), set `memory_enabled: false`. +> To turn memory off entirely (not just gate it), set both `memory_enabled: false` and `user_profile_enabled: false`. When both built-in stores are disabled, the built-in `memory` tool is automatically hidden. Review staged writes from the CLI or any messaging platform: From a92412ede151374212782b0bf47aa8df82e1c19d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 23:02:30 -0700 Subject: [PATCH 26/42] test: mock load_config_readonly in memory status gate tests check_memory_requirements() reads the readonly config loader; the salvaged tests only patched load_config, so the gate saw the real config. --- tests/hermes_cli/test_memory_setup.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/tests/hermes_cli/test_memory_setup.py b/tests/hermes_cli/test_memory_setup.py index 674a9b51d1..84bb7ec02d 100644 --- a/tests/hermes_cli/test_memory_setup.py +++ b/tests/hermes_cli/test_memory_setup.py @@ -100,9 +100,11 @@ def test_install_dependencies_force_reinstalls_versioned_specs(tmp_path, monkeyp def test_cmd_status_memory_tool_gate_disabled(capsys, monkeypatch): """When both memory stores are disabled, Memory status reports memory tool as disabled.""" + _cfg = {"memory": {"memory_enabled": False, "user_profile_enabled": False}} + monkeypatch.setattr("hermes_cli.config.load_config", lambda: _cfg) + # check_memory_requirements() reads the readonly loader, not load_config. monkeypatch.setattr( - "hermes_cli.config.load_config", - lambda: {"memory": {"memory_enabled": False, "user_profile_enabled": False}}, + "hermes_cli.config.load_config_readonly", lambda: _cfg, raising=False ) monkeypatch.setattr(memory_setup, "_get_available_providers", lambda: []) @@ -116,9 +118,10 @@ def test_cmd_status_memory_tool_gate_disabled(capsys, monkeypatch): def test_cmd_status_memory_tool_gate_enabled(capsys, monkeypatch): """When at least one memory store is enabled, Memory status reports memory tool as enabled.""" + _cfg = {"memory": {"memory_enabled": True, "user_profile_enabled": False}} + monkeypatch.setattr("hermes_cli.config.load_config", lambda: _cfg) monkeypatch.setattr( - "hermes_cli.config.load_config", - lambda: {"memory": {"memory_enabled": True, "user_profile_enabled": False}}, + "hermes_cli.config.load_config_readonly", lambda: _cfg, raising=False ) monkeypatch.setattr(memory_setup, "_get_available_providers", lambda: []) From 8436e0d142b8756dc9f16c79ec24da2f549ad392 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 23:02:37 -0700 Subject: [PATCH 27/42] chore: map contributor email for #87427 salvage --- contributors/emails/cassie@omg.lol | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/cassie@omg.lol diff --git a/contributors/emails/cassie@omg.lol b/contributors/emails/cassie@omg.lol new file mode 100644 index 0000000000..598c4864e5 --- /dev/null +++ b/contributors/emails/cassie@omg.lol @@ -0,0 +1 @@ +acgh213 From dbbd8937aec18e0739aec9ba58f1d24f6cb12698 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 22:59:30 -0700 Subject: [PATCH 28/42] docs(computer-use): document requesting the actual screenshot on chat surfaces MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to PR #90183 — computer_use now saves a bounded shareable copy of image captures, so attachment-capable surfaces (Telegram, Discord, Desktop) can deliver the real screenshot when the user asks. Documents the behavior, the 20-file cache bound, and the no-automatic-send rule. --- .../docs/user-guide/features/computer-use.md | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/website/docs/user-guide/features/computer-use.md b/website/docs/user-guide/features/computer-use.md index 7215899c66..54fc0f8012 100644 --- a/website/docs/user-guide/features/computer-use.md +++ b/website/docs/user-guide/features/computer-use.md @@ -284,6 +284,23 @@ the model substitutes the platform's idiomatic shortcut and app name): During all of this, your cursor stays wherever you left it and the email app never comes to front. +## Receiving the actual screenshot + +Screenshots taken during computer control are normally internal — they exist +so the model can see the screen, and the agent replies in text. But every +image capture also saves a bounded, shareable copy under Hermes' image cache +and reports its path, so on attachment-capable surfaces (Telegram, Discord, +Desktop, and other gateway platforms) you can simply ask: + +> *"Send me a screenshot of my screen."* + +and the agent delivers the real image as a native attachment, not just a +description. On the CLI there is no attachment channel, so the agent gives +you the saved file's path instead. + +Only the 20 most recent capture files are kept, and screenshots are never +sent automatically — only when you ask for one. + ## Provider compatibility | Provider | Vision? | Works? | Notes | From 118dbe871f4164d4f64d07349627b145ef5baa84 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Wed, 19 Aug 2026 11:02:20 +0800 Subject: [PATCH 29/42] fix(cli): map kitty CSI-u lock-bit variants so key combos survive NumLock kitty and ghostty OR the CapsLock (64) / NumLock (128) state into the CSI-u modifier parameter. With NumLock on, Ctrl+C arrives as ESC[99;133u (5 + 128) instead of ESC[99;5u; the alias table had no entry for it, so every key combo leaked as literal text like [127;133u (#89651). Install every CSI-u alias with the lock-bit variants (+64/+128/+192); the xterm modifyOtherKeys encoding never carries lock bits, so the ESC[27;N;CP~ form is left untouched. The Esc-key registration now covers modifier 1 as well (1+128=129 is a lone Esc with NumLock on). --- hermes_cli/pt_input_extras.py | 35 +++++++++--- tests/cli/test_modify_other_keys_aliases.py | 59 +++++++++++++++++++++ 2 files changed, 87 insertions(+), 7 deletions(-) diff --git a/hermes_cli/pt_input_extras.py b/hermes_cli/pt_input_extras.py index d2800b0299..dfdd04ae73 100644 --- a/hermes_cli/pt_input_extras.py +++ b/hermes_cli/pt_input_extras.py @@ -181,6 +181,11 @@ def install_modify_other_keys_aliases() -> int: Ctrl+Alt+Shift=8): normalized onto the same targets — Ctrl-bearing combos behave as the Ctrl key (Alt adds an ``Escape`` prefix), matching how dte/kakoune normalize these protocols. + * **Lock-bit variants**: every CSI-u mapping above is also installed + with the CapsLock (64) and NumLock (128) bits ORed into the modifier + parameter — kitty/ghostty include them while a lock is on, and + without the variants every key combo dies with the lock enabled + (``ESC[99;133u`` instead of ``ESC[99;5u``, #89651). * **Esc key**: ``ESC[27u`` / ``ESC[27;u`` (Kitty disambiguate mode reports Esc this way, #56684) → ``Keys.Escape``. * **Modified Enter/Tab/Backspace/Space**: Alt+Enter → the Alt+Enter @@ -244,15 +249,24 @@ def install_modify_other_keys_aliases() -> int: changed = 0 + # Kitty CSI-u encodes CapsLock/NumLock state as extra modifier bits + # (caps=64, num=128) ORed into the parameter: with NumLock on, Ctrl+C + # arrives as ESC[99;133u (5 + 128) instead of ESC[99;5u. Terminals + # that report these bits (kitty, ghostty) break every key combo while + # a lock is on (#89651) unless the lock variants are mapped too. The + # xterm modifyOtherKeys encoding never carries the lock bits, so only + # the CSI-u form needs them. + _CSI_U_LOCK_BIT_VARIANTS = (0, 64, 128, 192) + def _install_paired(modifier: int, mapping: dict) -> None: """Install both modifyOtherKeys (ESC[27;N;CP~) and CSI-u (ESC[CP;Nu) mappings for the given modifier and codepoint→key mapping.""" nonlocal changed for codepoint, key_val in mapping.items(): - for seq in ( - f"\x1b[27;{modifier};{codepoint}~", - f"\x1b[{codepoint};{modifier}u", - ): + seqs = [f"\x1b[27;{modifier};{codepoint}~"] + for lock_bits in _CSI_U_LOCK_BIT_VARIANTS: + seqs.append(f"\x1b[{codepoint};{modifier + lock_bits}u") + for seq in seqs: if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = key_val changed += 1 @@ -319,9 +333,16 @@ def install_modify_other_keys_aliases() -> int: # Disambiguate mode reports the Esc key as CSI-u so it is # distinguishable from the ESC byte that starts escape sequences # (#56684 — previously leaked "[27u" as literal text into the prompt). - # Modifiers run to 16 because kitty reports Cmd as the super bit - # (mod 9+) — same reason install_cmd_backspace_alias maps 9/10. - for seq in ["\x1b[27u"] + [f"\x1b[27;{m}u" for m in range(2, 17)]: + # Modifiers run from 1 to 16: kitty reports Cmd as the super bit + # (mod 9+) — same reason install_cmd_backspace_alias maps 9/10 — and + # the lock-bit variants of the modifier-less form (1+64/128/192) are + # how a lone Esc keypress arrives with a lock on. Lock bits (caps/num) + # get the same variant treatment as _install_paired. + for seq in ["\x1b[27u"] + [ + f"\x1b[27;{m + lock_bits}u" + for m in range(1, 17) + for lock_bits in _CSI_U_LOCK_BIT_VARIANTS + ]: if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = Keys.Escape changed += 1 diff --git a/tests/cli/test_modify_other_keys_aliases.py b/tests/cli/test_modify_other_keys_aliases.py index 040963722e..a4d35d918b 100644 --- a/tests/cli/test_modify_other_keys_aliases.py +++ b/tests/cli/test_modify_other_keys_aliases.py @@ -433,3 +433,62 @@ def test_cmd_backspace_alias_not_clobbered(): install_cmd_backspace_alias() install_modify_other_keys_aliases() assert _parse("\x1b[127;9u") == [Keys.ControlU] + + +# --------------------------------------------------------------------------- +# Lock-bit variants (#89651): kitty/ghostty OR the CapsLock (64) / NumLock +# (128) state into the CSI-u modifier parameter, so with a lock enabled +# every combo arrives shifted (ESC[99;133u instead of ESC[99;5u) and died +# as literal text without these aliases. +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize("letter", CTRL_LETTERS) +def test_ctrl_letter_with_numlock_parses_as_raw_byte(letter): + """Ctrl+ with NumLock on (modifier + 128) must parse identically + to the raw control byte — the exact garbage from #89651 ([127;133u).""" + raw_byte = chr(ord(letter) - ord('a') + 1) + raw_result = _parse(raw_byte) + + numlock_seq = f"\x1b[{ord(letter)};133u" # 5 + 128 + assert _parse(numlock_seq) == raw_result, ( + f"NumLock Ctrl+{letter} ({numlock_seq!r}) should parse identically " + f"to raw {raw_byte!r}" + ) + + +@pytest.mark.parametrize("letter", ["a", "c", "z"]) +def test_ctrl_letter_with_capslock_parses_as_raw_byte(letter): + raw_byte = chr(ord(letter) - ord('a') + 1) + capslock_seq = f"\x1b[{ord(letter)};69u" # 5 + 64 + assert _parse(capslock_seq) == _parse(raw_byte) + + +def test_ctrl_c_with_both_locks_parses_as_raw_byte(): + """Ctrl+C with CapsLock and NumLock both on (5 + 64 + 128 = 197).""" + assert _parse("\x1b[99;197u") == _parse("\x03") + + +def test_alt_letter_with_numlock_keeps_escape_prefix(): + assert _parse("\x1b[97;131u") == [Keys.Escape, "a"] # 3 + 128 + + +def test_shift_letter_with_capslock_types_uppercase(): + assert _parse("\x1b[97;66u") == ["A"] # 2 + 64 + + +def test_esc_key_with_numlock_is_escape(): + assert _parse("\x1b[27;129u") == [Keys.Escape] # 1 + 128 + assert _parse("\x1b[27;133u") == [Keys.Escape] # 5 + 128 + + +def test_ctrl_backspace_with_numlock_is_backward_kill_word(): + """The exact sequence from the #89651 report ([127;133u).""" + assert _parse("\x1b[127;133u") == [Keys.Escape, Keys.ControlH] + + +def test_modify_other_keys_tilde_form_has_no_lock_variants(): + """The xterm modifyOtherKeys encoding never carries lock bits, so no + +64/+128 variants of the ESC[27;N;CP~ form may be installed.""" + for seq in ("\x1b[27;69;99~", "\x1b[27;133;99~", "\x1b[27;197;99~"): + assert seq not in ANSI_SEQUENCES From 6471f53619f544480176af3308c5b27537ad202d Mon Sep 17 00:00:00 2001 From: grahf Date: Thu, 20 Aug 2026 07:30:20 +1000 Subject: [PATCH 30/42] fix(cli): map NumLock/CapsLock modifier-bit variants under kitty protocol With the kitty keyboard protocol push active (CSI >1u disambiguate), kitty encodes lock-key state into the CSI modifier field of function keys: a plain Down with NumLock on arrives as ESC[1;129B (NumLock), ESC[1;65B (CapsLock), or ESC[1;193B (both) instead of the legacy ESC[B. Stock prompt_toolkit maps none of these, so the parser fires Escape and inserts the remainder as literal text in the input line. 03bf85d83 restored the kitty push and completed the extended-key alias table but left these lock-bit variants unmapped, so kitty + NumLock still leaks [1;129A/B for arrow/nav keys. Map modifier 129/65/193 (plain) and 130/66/194 (+shift) for arrows, Home/End, Insert/Delete/PageUp/PageDown, and CSI-u Enter/Tab/Backspace/ Space to their plain keys. --- hermes_cli/pt_input_extras.py | 50 +++++++++++++++++++++++++++++++++++ 1 file changed, 50 insertions(+) diff --git a/hermes_cli/pt_input_extras.py b/hermes_cli/pt_input_extras.py index dfdd04ae73..ec3315fcad 100644 --- a/hermes_cli/pt_input_extras.py +++ b/hermes_cli/pt_input_extras.py @@ -366,6 +366,56 @@ def install_modify_other_keys_aliases() -> int: # matching Ink TUI + Desktop (#78285) }) + # -- Lock-key modifier bits (NumLock=128, CapsLock=64) under the kitty + # disambiguate push: kitty encodes lock state into the CSI sequence, so + # a plain Down with NumLock on arrives as ESC[1;129B (NumLock), ESC[1;65B + # (CapsLock) or ESC[1;193B (both) instead of the legacy ESC[B that stock + # prompt_toolkit maps. Those fall through the parser and leak as literal + # text ("[1;129B") in the input line. Map them to the plain key; the + # +shift variants (130/66/194) to the Shift* keys. + lock_plain = (129, 65, 193) + lock_shift = (130, 66, 194) + for mod in lock_plain: + for trailer, key in ( + ("A", Keys.Up), ("B", Keys.Down), ("C", Keys.Right), ("D", Keys.Left), + ("H", Keys.Home), ("F", Keys.End), + ): + seq = f"\x1b[1;{mod}{trailer}" + if seq not in ANSI_SEQUENCES: + ANSI_SEQUENCES[seq] = key + changed += 1 + for num, key in ( + (2, Keys.Insert), (3, Keys.Delete), (5, Keys.PageUp), (6, Keys.PageDown), + ): + seq = f"\x1b[{num};{mod}~" + if seq not in ANSI_SEQUENCES: + ANSI_SEQUENCES[seq] = key + changed += 1 + for cp, key in ( + (13, Keys.ControlM), # Enter + (9, Keys.Tab), # Tab + (127, Keys.Backspace), # Backspace + (32, " "), # Space + ): + seq = f"\x1b[{cp};{mod}u" + if seq not in ANSI_SEQUENCES: + ANSI_SEQUENCES[seq] = key + changed += 1 + for mod in lock_shift: + for trailer, key in ( + ("A", Keys.ShiftUp), ("B", Keys.ShiftDown), ("C", Keys.ShiftRight), + ("D", Keys.ShiftLeft), ("H", Keys.ShiftHome), ("F", Keys.ShiftEnd), + ): + seq = f"\x1b[1;{mod}{trailer}" + if seq not in ANSI_SEQUENCES: + ANSI_SEQUENCES[seq] = key + changed += 1 + for num, key in ((2, Keys.ShiftInsert), (3, Keys.ShiftDelete)): + seq = f"\x1b[{num};{mod}~" + if seq not in ANSI_SEQUENCES: + ANSI_SEQUENCES[seq] = key + changed += 1 + # -- Kitty functional keys (Private Use Area codepoints) ---- # kitty emits these CSI-u encodings even in LEGACY mode for keys that # have no legacy encoding, so unmapped they leak as literal text in any From 446da3ef56c99a974efb9c4a5c270bdf9f1849fe Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 20 Aug 2026 10:55:54 +0530 Subject: [PATCH 31/42] fix(cli): widen lock-bit coverage to alias installers, legacy nav keys, and PUA functional keys Follow-up to the salvaged #89676 + #90291 lock-bit fixes: extract a shared _lock_variants() helper and cover the sites both PRs missed - install_shift_enter_alias / install_ctrl_enter_alias / install_cmd_backspace_alias CSI-u spellings, legacy CSI-letter and CSI-tilde navigation twins derived from the existing table for ALL modifiers 1-16 (not just plain/shift), plain F1-F4 SS3 fallback, unmodified CSI-u keys (Tab/Enter/Space/Backspace), and kitty PUA functional keys (keypad, F13-F24, Ignore range) under lock bits. 8 new tests. --- contributors/emails/contact@grahfmusic.com | 1 + hermes_cli/pt_input_extras.py | 146 +++++++++++--------- tests/cli/test_modify_other_keys_aliases.py | 68 +++++++++ 3 files changed, 152 insertions(+), 63 deletions(-) create mode 100644 contributors/emails/contact@grahfmusic.com diff --git a/contributors/emails/contact@grahfmusic.com b/contributors/emails/contact@grahfmusic.com new file mode 100644 index 0000000000..6beb2eb18c --- /dev/null +++ b/contributors/emails/contact@grahfmusic.com @@ -0,0 +1 @@ +grahfmusic diff --git a/hermes_cli/pt_input_extras.py b/hermes_cli/pt_input_extras.py index ec3315fcad..d878bbee51 100644 --- a/hermes_cli/pt_input_extras.py +++ b/hermes_cli/pt_input_extras.py @@ -11,6 +11,19 @@ can be unit-tested without importing the whole CLI runtime. from __future__ import annotations +# kitty CSI-u ORs lock-key state into the modifier parameter of every key +# event while a lock is on: CapsLock=64, NumLock=128, both=192 (#88221, +# #89651). Every fixed-modifier CSI-u (and legacy CSI-tilde / CSI-letter) +# registration therefore needs lock-offset twins, or those events leak into +# the prompt as literal text. The xterm modifyOtherKeys ``ESC[27;N;CP~`` +# encoding never carries lock bits, so it never gets the twins. +_LOCK_BIT_OFFSETS = (0, 64, 128, 192) + + +def _lock_variants(modifier: int) -> tuple[int, ...]: + """Return ``modifier`` plus its CapsLock/NumLock/both twins.""" + return tuple(modifier + off for off in _LOCK_BIT_OFFSETS) + def _clear_vt100_prefix_cache() -> None: """Drop prompt_toolkit's memoized "is this a prefix of a longer match?" @@ -62,7 +75,9 @@ def install_shift_enter_alias() -> int: alt_enter = (Keys.Escape, Keys.ControlM) changed = 0 - for seq in ("\x1b[13;2u", "\x1b[27;2;13~", "\x1b[27;2;13u"): + seqs = [f"\x1b[13;{m}u" for m in _lock_variants(2)] + seqs += ["\x1b[27;2;13~", "\x1b[27;2;13u"] + for seq in seqs: if ANSI_SEQUENCES.get(seq) != alt_enter: ANSI_SEQUENCES[seq] = alt_enter changed += 1 @@ -96,7 +111,9 @@ def install_ctrl_enter_alias() -> int: alt_enter = (Keys.Escape, Keys.ControlM) changed = 0 - for seq in ("\x1b[13;5u", "\x1b[27;5;13~", "\x1b[27;5;13u"): + seqs = [f"\x1b[13;{m}u" for m in _lock_variants(5)] + seqs += ["\x1b[27;5;13~", "\x1b[27;5;13u"] + for seq in seqs: if ANSI_SEQUENCES.get(seq) != alt_enter: ANSI_SEQUENCES[seq] = alt_enter changed += 1 @@ -133,13 +150,12 @@ def install_cmd_backspace_alias() -> int: except Exception: return 0 - aliases = { - "\x1b[127;9u": Keys.ControlU, - "\x1b[127;10u": Keys.ControlU, - "\x1b[27;9;127~": Keys.ControlU, - "\x1b[3;9~": Keys.ControlK, - "\x1b[3;10~": Keys.ControlK, - } + aliases: dict[str, object] = {} + for base in (9, 10): # super / super+shift + for mod in _lock_variants(base): + aliases[f"\x1b[127;{mod}u"] = Keys.ControlU + aliases[f"\x1b[3;{mod}~"] = Keys.ControlK + aliases["\x1b[27;9;127~"] = Keys.ControlU changed = 0 for seq, key in aliases.items(): if ANSI_SEQUENCES.get(seq) != key: @@ -256,16 +272,14 @@ def install_modify_other_keys_aliases() -> int: # a lock is on (#89651) unless the lock variants are mapped too. The # xterm modifyOtherKeys encoding never carries the lock bits, so only # the CSI-u form needs them. - _CSI_U_LOCK_BIT_VARIANTS = (0, 64, 128, 192) - def _install_paired(modifier: int, mapping: dict) -> None: """Install both modifyOtherKeys (ESC[27;N;CP~) and CSI-u (ESC[CP;Nu) mappings for the given modifier and codepoint→key mapping.""" nonlocal changed for codepoint, key_val in mapping.items(): seqs = [f"\x1b[27;{modifier};{codepoint}~"] - for lock_bits in _CSI_U_LOCK_BIT_VARIANTS: - seqs.append(f"\x1b[{codepoint};{modifier + lock_bits}u") + for mod in _lock_variants(modifier): + seqs.append(f"\x1b[{codepoint};{mod}u") for seq in seqs: if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = key_val @@ -341,7 +355,7 @@ def install_modify_other_keys_aliases() -> int: for seq in ["\x1b[27u"] + [ f"\x1b[27;{m + lock_bits}u" for m in range(1, 17) - for lock_bits in _CSI_U_LOCK_BIT_VARIANTS + for lock_bits in _LOCK_BIT_OFFSETS ]: if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = Keys.Escape @@ -366,55 +380,55 @@ def install_modify_other_keys_aliases() -> int: # matching Ink TUI + Desktop (#78285) }) - # -- Lock-key modifier bits (NumLock=128, CapsLock=64) under the kitty - # disambiguate push: kitty encodes lock state into the CSI sequence, so - # a plain Down with NumLock on arrives as ESC[1;129B (NumLock), ESC[1;65B - # (CapsLock) or ESC[1;193B (both) instead of the legacy ESC[B that stock - # prompt_toolkit maps. Those fall through the parser and leak as literal - # text ("[1;129B") in the input line. Map them to the plain key; the - # +shift variants (130/66/194) to the Shift* keys. - lock_plain = (129, 65, 193) - lock_shift = (130, 66, 194) - for mod in lock_plain: - for trailer, key in ( - ("A", Keys.Up), ("B", Keys.Down), ("C", Keys.Right), ("D", Keys.Left), - ("H", Keys.Home), ("F", Keys.End), - ): - seq = f"\x1b[1;{mod}{trailer}" - if seq not in ANSI_SEQUENCES: - ANSI_SEQUENCES[seq] = key - changed += 1 - for num, key in ( - (2, Keys.Insert), (3, Keys.Delete), (5, Keys.PageUp), (6, Keys.PageDown), - ): - seq = f"\x1b[{num};{mod}~" - if seq not in ANSI_SEQUENCES: - ANSI_SEQUENCES[seq] = key - changed += 1 - for cp, key in ( - (13, Keys.ControlM), # Enter - (9, Keys.Tab), # Tab - (127, Keys.Backspace), # Backspace - (32, " "), # Space - ): - seq = f"\x1b[{cp};{mod}u" - if seq not in ANSI_SEQUENCES: - ANSI_SEQUENCES[seq] = key - changed += 1 - for mod in lock_shift: - for trailer, key in ( - ("A", Keys.ShiftUp), ("B", Keys.ShiftDown), ("C", Keys.ShiftRight), - ("D", Keys.ShiftLeft), ("H", Keys.ShiftHome), ("F", Keys.ShiftEnd), - ): - seq = f"\x1b[1;{mod}{trailer}" - if seq not in ANSI_SEQUENCES: - ANSI_SEQUENCES[seq] = key - changed += 1 - for num, key in ((2, Keys.ShiftInsert), (3, Keys.ShiftDelete)): - seq = f"\x1b[{num};{mod}~" - if seq not in ANSI_SEQUENCES: - ANSI_SEQUENCES[seq] = key - changed += 1 + # -- Unmodified keys with a lock bit set (kitty modifier 1 = "none") -- + # With a lock on, kitty stamps the lock bit onto keys pressed with NO + # real modifier too, so plain Backspace arrives as ESC[127;129u + # (1 + 128) rather than \x7f. _install_paired(1, ...) registers the + # bare mod-1 spelling and its lock twins. Only keys kitty CSI-u-encodes + # on their own are listed; plain text characters are still delivered + # as UTF-8, lock bits or not. + _install_paired(1, { + 9: Keys.ControlI, # Tab + 13: Keys.ControlM, # Enter + 32: " ", # Space + 127: Keys.ControlH, # Backspace + }) + + # -- Lock-key modifier bits (NumLock=128, CapsLock=64) on the legacy + # CSI-letter / CSI-tilde forms kitty keeps using under the disambiguate + # push: kitty encodes lock state into the modifier parameter, so a + # plain Down with NumLock on arrives as ESC[1;129B (NumLock), ESC[1;65B + # (CapsLock) or ESC[1;193B (both) instead of the legacy ESC[B — and a + # modified one shifts the same way (Alt+Left → ESC[1;131D). Those fall + # through the parser and leak as literal text ("[1;129B") in the input + # line. Derive the lock twins from whatever the table already maps for + # the base modifier (stock prompt_toolkit entries included), so every + # modifier the terminal can report keeps working under a lock. + for m in range(1, 17): + for lock in _LOCK_BIT_OFFSETS[1:]: + # CSI-letter navigation: Up/Down/Right/Left/End/Home + F1-F4 + for trailer in "ABCDFHPQRS": + base_seq = f"\x1b[1;{m}{trailer}" if m > 1 else f"\x1b[{trailer}" + key = ANSI_SEQUENCES.get(base_seq) + if key is None and m == 1: + # Plain F1-F4 live in the table as SS3 (ESC O P) forms. + key = ANSI_SEQUENCES.get(f"\x1bO{trailer}") + if key is None: + continue + seq = f"\x1b[1;{m + lock}{trailer}" + if seq not in ANSI_SEQUENCES: + ANSI_SEQUENCES[seq] = key + changed += 1 + # CSI-tilde navigation: Insert/Delete/PageUp/PageDown/Home/End + for num in (1, 2, 3, 4, 5, 6, 7, 8): + base_seq = f"\x1b[{num};{m}~" if m > 1 else f"\x1b[{num}~" + key = ANSI_SEQUENCES.get(base_seq) + if key is None: + continue + seq = f"\x1b[{num};{m + lock}~" + if seq not in ANSI_SEQUENCES: + ANSI_SEQUENCES[seq] = key + changed += 1 # -- Kitty functional keys (Private Use Area codepoints) ---- # kitty emits these CSI-u encodings even in LEGACY mode for keys that @@ -450,6 +464,12 @@ def install_modify_other_keys_aliases() -> int: if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = key_val changed += 1 + # Lock twins: with a lock on these arrive as ESC[;129u etc. + for mod in _lock_variants(1)[1:]: + seq = f"\x1b[{code};{mod}u" + if seq not in ANSI_SEQUENCES: + ANSI_SEQUENCES[seq] = key_val + changed += 1 # New longer sequences can flip "is this a prefix of a longer match?" # answers the VT100 parser already cached — drop the cache so parsers diff --git a/tests/cli/test_modify_other_keys_aliases.py b/tests/cli/test_modify_other_keys_aliases.py index a4d35d918b..36e24fd0ae 100644 --- a/tests/cli/test_modify_other_keys_aliases.py +++ b/tests/cli/test_modify_other_keys_aliases.py @@ -492,3 +492,71 @@ def test_modify_other_keys_tilde_form_has_no_lock_variants(): +64/+128 variants of the ESC[27;N;CP~ form may be installed.""" for seq in ("\x1b[27;69;99~", "\x1b[27;133;99~", "\x1b[27;197;99~"): assert seq not in ANSI_SEQUENCES + + +# --------------------------------------------------------------------------- +# Follow-up widening: lock twins on the alias installers, legacy CSI-letter / +# CSI-tilde navigation, unmodified CSI-u keys, and PUA functional keys. +# --------------------------------------------------------------------------- + + +def test_lock_bits_on_legacy_cursor_keys_map_to_plain_keys(): + """kitty stamps lock bits onto legacy CSI-letter arrows too: + plain Down + NumLock = ESC[1;129B, CapsLock = ESC[1;65B, both = 193.""" + for mod, key in ((129, Keys.Down), (65, Keys.Down), (193, Keys.Down)): + assert _parse(f"\x1b[1;{mod}B") == [key] + assert _parse("\x1b[1;130B") == [Keys.ShiftDown] # Shift + NumLock + assert _parse("\x1b[1;131D") == [Keys.Escape, Keys.Left] # Alt + NumLock + assert _parse("\x1b[1;133D") == [Keys.ControlLeft] # Ctrl + NumLock + + +def test_lock_bits_on_tilde_navigation_keys(): + """Delete/PageUp/etc. carry the modifier in CSI-tilde form.""" + assert _parse("\x1b[3;129~") == [Keys.Delete] + assert _parse("\x1b[3;69~") == [Keys.ControlDelete] # Ctrl+Delete + Caps + assert _parse("\x1b[5;193~") == [Keys.PageUp] # both locks + + +def test_lock_bits_on_plain_f1_through_f4(): + """Plain F1-F4 base mappings are SS3 (ESC O P); their CSI lock twins + must still resolve (ESC[1;129P etc.).""" + base = _parse("\x1bOP") + assert _parse("\x1b[1;129P") == base + assert _parse("\x1b[1;65P") == base + + +def test_lock_bits_on_unmodified_csi_u_keys(): + """Tab/Enter/Space/Backspace with only a lock held (modifier 1+lock).""" + assert _parse("\x1b[9;65u") == _parse("\t") + assert _parse("\x1b[13;193u") == _parse("\r") + assert _parse("\x1b[32;129u") == _parse(" ") + assert _parse("\x1b[127;129u") == _parse("\x7f") + + +def test_lock_bits_on_pua_functional_keys(): + """Kitty PUA functional keys (keypad, F13+) keep working under locks — + NumLock especially matters because it gates the keypad itself.""" + assert _parse("\x1b[57399;129u") == ["0"] # KP_0 + NumLock + assert _parse("\x1b[57376;129u") == [Keys.F13] # F13 + NumLock + assert _parse("\x1b[57427;129u") == [Keys.Ignore] # KP_BEGIN + NumLock + + +def test_lock_bits_on_shift_enter_and_ctrl_enter_aliases(): + from hermes_cli.pt_input_extras import ( + install_ctrl_enter_alias, + install_shift_enter_alias, + ) + install_shift_enter_alias() + install_ctrl_enter_alias() + newline = _parse("\x1b\r") + assert _parse("\x1b[13;130u") == newline # Shift+Enter + NumLock + assert _parse("\x1b[13;66u") == newline # Shift+Enter + CapsLock + assert _parse("\x1b[13;133u") == newline # Ctrl+Enter + NumLock + + +def test_lock_bits_on_cmd_backspace_alias(): + from hermes_cli.pt_input_extras import install_cmd_backspace_alias + install_cmd_backspace_alias() + assert _parse("\x1b[127;137u") == [Keys.ControlU] # Cmd+Backspace + NumLock + assert _parse("\x1b[127;73u") == [Keys.ControlU] # Cmd+Backspace + Caps + assert _parse("\x1b[3;137~") == [Keys.ControlK] # Cmd+FwdDel + NumLock From 9ed06ca2b655da00c9b67dc91e90d5fef77aa076 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 20 Aug 2026 11:48:41 +0530 Subject: [PATCH 32/42] refactor(cli): fold lock-bit review findings - docstrings, dead tilde forms, _lock_twins idiom Post-review cleanup on the salvage: update the three alias-installer docstrings to mention lock twins (and fix ctrl-enter's stale 'stock maps none of these' claim - stock maps the tilde form to plain ControlM, which the overwrite fixes); skip the never-emitted modifier-1 tilde forms in _install_paired; name the twins-only idiom as _lock_twins() and use it at the legacy-nav + PUA sites; route the Esc loop through _lock_variants; hoist the per-base table lookup out of the lock loop in the legacy-nav section. --- hermes_cli/pt_input_extras.py | 65 +++++++++++++++++++++-------------- 1 file changed, 40 insertions(+), 25 deletions(-) diff --git a/hermes_cli/pt_input_extras.py b/hermes_cli/pt_input_extras.py index d878bbee51..5f7fa6e931 100644 --- a/hermes_cli/pt_input_extras.py +++ b/hermes_cli/pt_input_extras.py @@ -25,6 +25,11 @@ def _lock_variants(modifier: int) -> tuple[int, ...]: return tuple(modifier + off for off in _LOCK_BIT_OFFSETS) +def _lock_twins(modifier: int) -> tuple[int, ...]: + """Return only the lock twins of ``modifier`` (never the base value).""" + return tuple(modifier + off for off in _LOCK_BIT_OFFSETS[1:]) + + def _clear_vt100_prefix_cache() -> None: """Drop prompt_toolkit's memoized "is this a prefix of a longer match?" answers after mutating ``ANSI_SEQUENCES``. @@ -50,6 +55,7 @@ def install_shift_enter_alias() -> int: Sequences mapped: - "\\x1b[13;2u" — Kitty keyboard protocol / CSI-u, modifier=2 (Shift) + (plus its CapsLock/NumLock lock twins via ``_lock_variants``) - "\\x1b[27;2;13~" — xterm modifyOtherKeys=2, modifier=2 (Shift) - "\\x1b[27;2;13u" — alternate ordering some emitters use @@ -93,10 +99,13 @@ def install_ctrl_enter_alias() -> int: Sequences mapped: - "\\x1b[13;5u" — Kitty keyboard protocol / CSI-u, modifier=5 (Ctrl) + (plus its CapsLock/NumLock lock twins via ``_lock_variants``) - "\\x1b[27;5;13~" — xterm modifyOtherKeys=2, modifier=5 (Ctrl) - "\\x1b[27;5;13u" — alternate ordering some emitters use - Stock prompt_toolkit doesn't map any of these. Without this alias, + Stock prompt_toolkit maps only the tilde form ``\\x1b[27;5;13~`` (to + plain ``Keys.ControlM``, which this deliberately overwrites — same + bug-fix rationale as install_shift_enter_alias). Without this alias, Kitty/mintty/xterm-with-modifyOtherKeys users over SSH never get a Ctrl+Enter newline — the keystroke arrives as a raw CSI sequence that falls through to the default character-insert handler. See #22379. @@ -133,7 +142,8 @@ def install_cmd_backspace_alias() -> int: literal insertion. Cmd+Backspace → ``Keys.ControlU`` (kill backward to start of line). - Codepoint 127 with modifier 9 (super) / 10 (super+shift): + Codepoint 127 with modifier 9 (super) / 10 (super+shift), each with + its CapsLock/NumLock lock twins via ``_lock_variants``: - ``\\x1b[127;9u`` / ``\\x1b[127;10u`` — Kitty CSI-u - ``\\x1b[27;9;127~`` — xterm modifyOtherKeys @@ -274,10 +284,14 @@ def install_modify_other_keys_aliases() -> int: # the CSI-u form needs them. def _install_paired(modifier: int, mapping: dict) -> None: """Install both modifyOtherKeys (ESC[27;N;CP~) and CSI-u (ESC[CP;Nu) - mappings for the given modifier and codepoint→key mapping.""" + mappings for the given modifier and codepoint→key mapping. + + The tilde form is skipped for modifier 1 ("no modifier") — xterm + never emits modifier-1 tilde sequences. + """ nonlocal changed for codepoint, key_val in mapping.items(): - seqs = [f"\x1b[27;{modifier};{codepoint}~"] + seqs = [] if modifier == 1 else [f"\x1b[27;{modifier};{codepoint}~"] for mod in _lock_variants(modifier): seqs.append(f"\x1b[{codepoint};{mod}u") for seq in seqs: @@ -353,9 +367,9 @@ def install_modify_other_keys_aliases() -> int: # how a lone Esc keypress arrives with a lock on. Lock bits (caps/num) # get the same variant treatment as _install_paired. for seq in ["\x1b[27u"] + [ - f"\x1b[27;{m + lock_bits}u" + f"\x1b[27;{mod}u" for m in range(1, 17) - for lock_bits in _LOCK_BIT_OFFSETS + for mod in _lock_variants(m) ]: if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = Keys.Escape @@ -405,27 +419,28 @@ def install_modify_other_keys_aliases() -> int: # the base modifier (stock prompt_toolkit entries included), so every # modifier the terminal can report keeps working under a lock. for m in range(1, 17): - for lock in _LOCK_BIT_OFFSETS[1:]: - # CSI-letter navigation: Up/Down/Right/Left/End/Home + F1-F4 - for trailer in "ABCDFHPQRS": - base_seq = f"\x1b[1;{m}{trailer}" if m > 1 else f"\x1b[{trailer}" - key = ANSI_SEQUENCES.get(base_seq) - if key is None and m == 1: - # Plain F1-F4 live in the table as SS3 (ESC O P) forms. - key = ANSI_SEQUENCES.get(f"\x1bO{trailer}") - if key is None: - continue - seq = f"\x1b[1;{m + lock}{trailer}" + # CSI-letter navigation: Up/Down/Right/Left/End/Home + F1-F4 + for trailer in "ABCDFHPQRS": + base_seq = f"\x1b[1;{m}{trailer}" if m > 1 else f"\x1b[{trailer}" + key = ANSI_SEQUENCES.get(base_seq) + if key is None and m == 1: + # Plain F1-F4 live in the table as SS3 (ESC O P) forms. + key = ANSI_SEQUENCES.get(f"\x1bO{trailer}") + if key is None: + continue + for mod in _lock_twins(m): + seq = f"\x1b[1;{mod}{trailer}" if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = key changed += 1 - # CSI-tilde navigation: Insert/Delete/PageUp/PageDown/Home/End - for num in (1, 2, 3, 4, 5, 6, 7, 8): - base_seq = f"\x1b[{num};{m}~" if m > 1 else f"\x1b[{num}~" - key = ANSI_SEQUENCES.get(base_seq) - if key is None: - continue - seq = f"\x1b[{num};{m + lock}~" + # CSI-tilde navigation: Insert/Delete/PageUp/PageDown/Home/End + for num in (1, 2, 3, 4, 5, 6, 7, 8): + base_seq = f"\x1b[{num};{m}~" if m > 1 else f"\x1b[{num}~" + key = ANSI_SEQUENCES.get(base_seq) + if key is None: + continue + for mod in _lock_twins(m): + seq = f"\x1b[{num};{mod}~" if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = key changed += 1 @@ -465,7 +480,7 @@ def install_modify_other_keys_aliases() -> int: ANSI_SEQUENCES[seq] = key_val changed += 1 # Lock twins: with a lock on these arrive as ESC[;129u etc. - for mod in _lock_variants(1)[1:]: + for mod in _lock_twins(1): seq = f"\x1b[{code};{mod}u" if seq not in ANSI_SEQUENCES: ANSI_SEQUENCES[seq] = key_val From f309f92d30474cee54ba1e229254de4db1824520 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 23:33:54 -0700 Subject: [PATCH 33/42] =?UTF-8?q?feat:=20hermes=20worktree=20list/prune=20?= =?UTF-8?q?=E2=80=94=20attended=20reclaim=20for=20accumulated=20worktrees?= =?UTF-8?q?=20and=20merged=20branches?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The startup pruner is deliberately conservative (unattended, pre-banner), so real installs accumulate what it can never touch: trees preserved for untracked-only scratch, and orphaned local branches beyond the two auto-generated prefixes it deletes. A measured multi-agent box: 35 trees / 15GB / 244 local branches, 120 of them fully merged. New attended surface (hermes_cli/worktree_gc.py + worktree_cmd.py): - hermes worktree list — audit every tree: age, size, verdict, reason, plus deletable-branch count - hermes worktree prune [--dry-run|--trees-only|--branches-only] - /worktree prune [--dry-run] — same engine in-session; never touches the session's own active tree - startup escalation: one WARNING when .worktrees/ exceeds 10 trees or 5GB, naming the reclaim commands (silence is how boxes hit 15GB) Safety invariants (shared with the startup pruner via cli.py primitives): tracked modifications and unique unpushed commits never deleted at any age; live-locked trees untouched; branch deletion gated on worktree removal success; untracked-only scratch ARCHIVED to ~/.hermes/archive/worktree-prune/ before its tree is reaped. Branch GC is content-gated, not name-gated: any local branch fully merged or git-cherry patch-equivalent upstream is safe to delete (rebase merges rewrite SHAs, so --merged alone misses the dominant leak); unique-commit, checked-out, protected, and stale-base (>50 ahead) branches are kept. Classification is parallel (8 workers) — 244 branches audit in ~64s live. git timeouts degrade to keep (returncode 124) instead of crashing the audit — live-verified failure on a 746MB .git repo. 16 behavior-contract tests against real git fixtures; live dry-run on the production repo: 12 trees reclaimable, 120 branches deletable, 0 false positives among kept trees. --- cli.py | 21 +- hermes_cli/cli_commands_mixin.py | 53 +++- hermes_cli/commands.py | 6 +- hermes_cli/main.py | 53 +++- hermes_cli/worktree_cmd.py | 99 ++++++ hermes_cli/worktree_gc.py | 432 +++++++++++++++++++++++++++ tests/hermes_cli/test_worktree_gc.py | 243 +++++++++++++++ website/docs/user-guide/cli.md | 37 +++ 8 files changed, 935 insertions(+), 9 deletions(-) create mode 100644 hermes_cli/worktree_cmd.py create mode 100644 hermes_cli/worktree_gc.py create mode 100644 tests/hermes_cli/test_worktree_gc.py diff --git a/cli.py b/cli.py index 329f6efa7e..e026a56138 100644 --- a/cli.py +++ b/cli.py @@ -2741,12 +2741,31 @@ def _prune_stale_worktrees(repo_root: str, max_age_hours: int = 24) -> None: if preserved_stale: logger.warning( "Preserving %d worktree(s) older than 7 days with unmerged work " - "(push or remove them to reclaim disk): %s", + "(run `hermes worktree prune` to review and reclaim): %s", len(preserved_stale), ", ".join(sorted(preserved_stale)), ) _prune_orphaned_branches(repo_root) + # Escalation notice: the startup pass is deliberately conservative, so + # installs accumulate preserved trees it can never reclaim. Once the + # footprint is clearly a problem (many trees or multi-GB), say so once + # per launch and name the attended reclaim command — silence here is how + # boxes reach 15GB+ of .worktrees/ without anyone noticing. + try: + from hermes_cli.worktree_gc import worktrees_summary + + count, size_mb = worktrees_summary(repo_root) + if count >= 10 or (size_mb or 0) >= 5120: + size_txt = f"{size_mb / 1024:.1f}GB" if size_mb else "unknown size" + logger.warning( + ".worktrees/ holds %d tree(s) (%s) — run `hermes worktree list` " + "to audit and `hermes worktree prune` to reclaim safely.", + count, size_txt, + ) + except Exception: + pass + def _prune_orphaned_branches(repo_root: str) -> None: """Delete local ``hermes/hermes-*`` and ``pr-*`` branches with no worktree. diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py index 1778c28749..f3dcd7399f 100644 --- a/hermes_cli/cli_commands_mixin.py +++ b/hermes_cli/cli_commands_mixin.py @@ -1219,18 +1219,23 @@ class CLICommandsMixin: self._handle_resume_command(f"/resume {arg}") def _handle_worktree_command(self, cmd_original: str) -> None: - """Handle /worktree — inspect or create isolated git worktrees. + """Handle /worktree — inspect, create, or reclaim isolated git worktrees. Syntax: - /worktree — show the active worktree (if any) - /worktree new [name] — create a worktree and move this session into it - /worktree list — list worktrees under the repo's .worktrees/ + /worktree — show the active worktree (if any) + /worktree new [name] — create a worktree and move this session into it + /worktree list — list worktrees under the repo's .worktrees/ + /worktree prune [--dry-run] — reclaim safe trees + merged branches Inspired by Copilot CLI's ``/worktree new``: start isolated work in a fresh worktree without leaving the session. Creating one retargets the terminal/file tools (``TERMINAL_CWD`` + process cwd) at the new tree; the launcher's exit cleanup applies (kept only when it has unpushed commits, same as ``hermes -w``). + + ``prune`` is the same attended reclaim as ``hermes worktree prune`` + (hermes_cli/worktree_gc.py): never deletes tracked changes, unique + unpushed commits, or in-use trees; archives untracked-only scratch. """ import subprocess @@ -1250,10 +1255,50 @@ class CLICommandsMixin: print(" No active worktree for this session.") if repo_root: print(" /worktree new [name] — create one and move this session into it") + print(" /worktree prune — reclaim stale trees and merged branches") else: print(" (not inside a git repository)") return + if sub in {"prune", "gc", "clean"}: + if not repo_root: + print(" Not inside a git repository.") + return + rest = parts[2].strip().lower() if len(parts) > 2 else "" + dry_run = "--dry-run" in rest or "-n" in rest.split() + from hermes_cli import worktree_gc + + active = _cli._active_worktree + tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False) + if active: + # Never reap the tree this very session is sitting in, even + # if a concurrent audit would judge it clean+merged. + active_path = str(active.get("path") or "") + tree_records = [ + record for record in tree_records + if record.path != active_path + ] + actions = worktree_gc.reclaim_worktrees( + repo_root, dry_run=dry_run, records=tree_records + ) + actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run) + if actions: + for line in actions: + print(f" {line}") + print(f" {len(actions)} action(s) {'planned' if dry_run else 'done'}.") + else: + print(" Nothing to reclaim — remaining trees/branches carry real work.") + kept = [ + record for record in tree_records + if record.verdict == "keep" + and "kanban" not in record.reason and "in use" not in record.reason + ] + if kept: + print(f" Preserved {len(kept)} tree(s) with real work:") + for record in kept: + print(f" {record.name}: {record.reason}") + return + if sub in {"list", "ls"}: if not repo_root: print(" Not inside a git repository.") diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py index 350ee7411c..ee8eda30a4 100644 --- a/hermes_cli/commands.py +++ b/hermes_cli/commands.py @@ -167,9 +167,9 @@ COMMAND_REGISTRY: list[CommandDef] = [ args_hint="", cli_only=True), CommandDef("branch", "Branch the current session (explore a different path)", "Session", aliases=("fork",), args_hint="[name]"), - CommandDef("worktree", "Show, list, or create isolated git worktrees for this session", "Session", - cli_only=True, args_hint="[new [name]|list]", - subcommands=("new", "list")), + CommandDef("worktree", "Show, list, create, or prune isolated git worktrees", "Session", + cli_only=True, args_hint="[new [name]|list|prune [--dry-run]]", + subcommands=("new", "list", "prune")), CommandDef("compress", "Compress conversation context (add 'here [N]' to keep recent N turns; --preview shows what would happen)", "Session", aliases=("compact",), args_hint="[here [N] | focus topic | --preview|--dry-run]"), CommandDef("rollback", "List or restore filesystem checkpoints (restores keep your hand-edits; --all overrides)", "Session", diff --git a/hermes_cli/main.py b/hermes_cli/main.py index c9025d4d9c..0a19ed3048 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -11644,7 +11644,7 @@ _BUILTIN_SUBCOMMANDS = frozenset( "resume", "send", "sessions", "setup", "skin", "skills", "slack", "status", "sync", "tools", "uninstall", "update", - "version", "webhook", "whatsapp", "whatsapp-cloud", "chat", "secrets", "security", + "version", "webhook", "whatsapp", "whatsapp-cloud", "worktree", "chat", "secrets", "security", "verify", # Help-ish invocations — plugin commands not being listed in # top-level --help is an acceptable trade-off for skipping an @@ -12514,6 +12514,57 @@ def main(): ) fallback_parser.set_defaults(func=cmd_fallback) + # ========================================================================= + # worktree command — audit/reclaim accumulated git worktrees + branches + # ========================================================================= + worktree_parser = subparsers.add_parser( + "worktree", + help="Audit and reclaim accumulated git worktrees and merged branches", + description=( + "Attended reclaim for the .worktrees/ directory hermes -w sessions " + "accumulate. Never deletes uncommitted tracked changes, unique " + "unpushed commits, or in-use trees; untracked-only scratch is " + "archived to ~/.hermes/archive/worktree-prune/ before removal. See: " + "https://hermes-agent.nousresearch.com/docs/user-guide/cli#worktree-cleanup" + ), + ) + worktree_subparsers = worktree_parser.add_subparsers(dest="worktree_action") + worktree_list = worktree_subparsers.add_parser( + "list", + aliases=["ls", "audit"], + help="Classify every tree: age, size, verdict, reason (default action)", + ) + worktree_list.add_argument("--repo", help="Repo root (default: current repo)") + worktree_prune = worktree_subparsers.add_parser( + "prune", + help="Remove safe trees and delete fully-merged local branches", + ) + worktree_prune.add_argument("--repo", help="Repo root (default: current repo)") + worktree_prune.add_argument( + "--dry-run", action="store_true", + help="Show the plan without changing anything", + ) + worktree_prune.add_argument( + "--trees-only", action="store_true", + help="Only remove worktrees; leave local branches alone", + ) + worktree_prune.add_argument( + "--branches-only", action="store_true", + help="Only delete merged local branches; leave worktrees alone", + ) + + def _dispatch_worktree(_args): + from hermes_cli.worktree_cmd import cmd_worktree + + # argparse aliases set dest to the literal typed string ("ls"/"audit"). + action = getattr(_args, "worktree_action", None) + if action in ("ls", "audit"): + _args.worktree_action = "list" + return cmd_worktree(_args) + + worktree_parser.set_defaults(func=_dispatch_worktree) + + # ========================================================================= # secrets command — external secret managers (Bitwarden, 1Password) # ========================================================================= diff --git a/hermes_cli/worktree_cmd.py b/hermes_cli/worktree_cmd.py new file mode 100644 index 0000000000..d615fe9fc3 --- /dev/null +++ b/hermes_cli/worktree_cmd.py @@ -0,0 +1,99 @@ +"""``hermes worktree`` — audit and reclaim accumulated git worktrees/branches. + +Attended counterpart of the silent startup pruner (see +``hermes_cli/worktree_gc.py`` for the policy and shared invariants). Usage: + + hermes worktree list # audit: verdict + reason per tree + hermes worktree prune # reap safe trees + merged branches + hermes worktree prune --dry-run # show the plan, change nothing + hermes worktree prune --trees-only / --branches-only +""" + +from __future__ import annotations + +from typing import Optional + + +def _repo_root() -> Optional[str]: + import cli as _cli + + return _cli._git_repo_root() + + +def _fmt_size(size_mb: Optional[int]) -> str: + if size_mb is None: + return "?" + if size_mb >= 1024: + return f"{size_mb / 1024:.1f}G" + return f"{size_mb}M" + + +def cmd_worktree(args) -> int: + from hermes_cli import worktree_gc + + repo_root = getattr(args, "repo", None) or _repo_root() + if not repo_root: + print("Not inside a git repository (or pass --repo ).") + return 1 + + action = getattr(args, "worktree_action", None) or "list" + + if action == "list": + records = worktree_gc.audit_worktrees(repo_root) + if not records: + print("No worktrees under .worktrees/ — nothing to reclaim.") + return 0 + total_mb = sum(r.size_mb or 0 for r in records) + reapable_mb = sum( + r.size_mb or 0 for r in records if r.verdict.startswith("reap") + ) + print(f"{'TREE':32} {'AGE':>6} {'SIZE':>6} {'VERDICT':13} REASON") + for r in sorted(records, key=lambda x: -(x.size_mb or 0)): + print( + f"{r.name[:32]:32} {r.age_days:>5.1f}d {_fmt_size(r.size_mb):>6} " + f"{r.verdict:13} {r.reason}" + ) + print( + f"\n{len(records)} tree(s), {_fmt_size(total_mb)} total — " + f"{_fmt_size(reapable_mb)} reclaimable now via `hermes worktree prune`." + ) + branch_records = worktree_gc.audit_branches(repo_root) + deletable = [b for b in branch_records if b.verdict == "delete"] + if deletable: + print( + f"{len(deletable)} local branch(es) fully merged/patch-equivalent " + f"upstream would also be deleted." + ) + return 0 + + if action == "prune": + dry_run = bool(getattr(args, "dry_run", False)) + trees_only = bool(getattr(args, "trees_only", False)) + branches_only = bool(getattr(args, "branches_only", False)) + + actions: list = [] + if not branches_only: + tree_records = worktree_gc.audit_worktrees(repo_root, with_sizes=False) + actions += worktree_gc.reclaim_worktrees( + repo_root, dry_run=dry_run, records=tree_records + ) + kept = [r for r in tree_records if r.verdict == "keep" + and "kanban" not in r.reason and "in use" not in r.reason] + if kept: + print(f"Preserved {len(kept)} tree(s) with real work:") + for r in kept: + print(f" {r.name}: {r.reason}") + if not trees_only: + actions += worktree_gc.reclaim_branches(repo_root, dry_run=dry_run) + + if actions: + for line in actions: + print(f" {line}") + verb = "planned" if dry_run else "done" + print(f"{len(actions)} action(s) {verb}.") + else: + print("Nothing to reclaim — all trees/branches carry real work or are in use.") + return 0 + + print(f"Unknown worktree action: {action}") + return 1 diff --git a/hermes_cli/worktree_gc.py b/hermes_cli/worktree_gc.py new file mode 100644 index 0000000000..bfb863c869 --- /dev/null +++ b/hermes_cli/worktree_gc.py @@ -0,0 +1,432 @@ +"""On-demand worktree + branch reclaim (``hermes worktree`` / ``/worktree prune``). + +The startup pruner in ``cli._prune_stale_worktrees`` is deliberately +conservative and silent: it runs before the banner on every ``hermes -w`` +launch, so it only reaps clean, fully-merged scratch trees past an age tier +and preserves everything else. That policy is correct for an unattended +startup path — but it means real installs accumulate two kinds of debris the +startup pass can never touch: + +- **Preserved trees** whose only "dirt" is untracked scratch (PR body drafts, + logs) on an otherwise merged branch — preserved forever by the dirty guard. +- **Orphaned local branches** beyond the two auto-generated prefixes the + startup pass deletes (``hermes/hermes-*``, ``pr-*``): salvage lanes, port + branches, feature branches whose PRs merged months ago. Multi-agent boxes + reach hundreds. + +This module is the *attended* counterpart: an explicit, loud, dry-run-first +reclaim the user invokes, so it can be more thorough while staying just as +safe. Invariants shared with the startup pruner (never violated here either): + +- tracked modifications are NEVER deleted, at any age, in any mode; +- unique unpushed commits are NEVER deleted (``git cherry`` patch-equivalence + decides "unique"; shallow repos are deepened bloblessly first so the + verdict is trustworthy); +- live-locked trees (owning pid alive) are never touched; +- a branch is deleted only after its worktree removal succeeded — a failed + removal must not orphan reachable commits; +- untracked-only dirt is ARCHIVED to ``~/.hermes/archive/worktree-prune/`` + before its tree is reaped, never destroyed. + +Classification primitives are imported from ``cli`` so the two paths can +never drift apart on what "dirty", "unpushed", or "merged" means. +""" + +from __future__ import annotations + +import logging +import os +import re +import shutil +import subprocess +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import List, Optional + +logger = logging.getLogger(__name__) + +# Branches never considered for deletion, in any mode. +_PROTECTED_BRANCHES = {"main", "master", "develop", "dev", "trunk"} + +# Trees owned by another lifecycle (kanban dispatcher gc) — never touched. +_KANBAN_RE = re.compile(r"^t_[0-9a-f]+$") + +# Bounded cherry probe: a branch this far ahead of upstream is a stale-base +# lane, not merged scratch; checking it is expensive and it stays preserved. +_MAX_CHERRY_AHEAD = 50 + + +@dataclass +class TreeRecord: + name: str + path: str + branch: str + age_days: float + size_mb: Optional[int] + verdict: str # reap | reap-archive | keep + reason: str + untracked: List[str] = field(default_factory=list) + + +@dataclass +class BranchRecord: + name: str + verdict: str # delete | keep + reason: str + + +def _git(args: list, cwd: str, timeout: int = 15) -> subprocess.CompletedProcess: + """Run git, translating timeouts into a nonzero returncode. + + Every verdict in this module fails safe toward "keep" on a nonzero + returncode, so a hung/slow git call (large repos make ``git cherry`` + genuinely slow) must degrade to keep — never crash the whole audit + (live-verified failure on a 746MB .git: TimeoutExpired escaped and + aborted the branch audit mid-list). + """ + try: + return subprocess.run( + ["git", *args], + capture_output=True, text=True, encoding="utf-8", + errors="replace", timeout=timeout, cwd=cwd, + ) + except subprocess.TimeoutExpired: + return subprocess.CompletedProcess( + args=["git", *args], returncode=124, + stdout="", stderr=f"timeout after {timeout}s", + ) + + +def _tree_size_mb(path: Path) -> Optional[int]: + """Cheap directory size via ``du -sm`` — best-effort, None on failure.""" + try: + result = subprocess.run( + ["du", "-sm", str(path)], + capture_output=True, text=True, encoding="utf-8", + errors="replace", timeout=30, + ) + if result.returncode == 0 and result.stdout.strip(): + return int(result.stdout.split()[0]) + except Exception: + pass + return None + + +def _dirty_split(path: str) -> tuple[bool, List[str]]: + """Return (has_tracked_modifications, untracked_paths). + + ``git status --porcelain`` counts untracked scratch equally with real + edits; the reclaim policy treats them very differently (tracked = real + work, untracked = archivable scratch), so split here. + """ + try: + result = _git(["status", "--porcelain"], cwd=path, timeout=10) + if result.returncode != 0: + return True, [] # fail safe: treat as real work + tracked = False + untracked: List[str] = [] + for line in result.stdout.splitlines(): + if not line.strip(): + continue + if line.startswith("??"): + untracked.append(line[3:].strip()) + else: + tracked = True + return tracked, untracked + except Exception: + return True, [] + + +def _archive_untracked(tree: Path, untracked: List[str]) -> Optional[Path]: + """Copy untracked files out of a doomed tree. Returns the archive dir. + + Never destroys: on any copy failure the caller must treat the tree as + keep. Costs almost nothing and removes the "did I just delete + something?" question. + """ + stamp = time.strftime("%Y%m%d-%H%M%S") + dest = ( + Path.home() / ".hermes" / "archive" / "worktree-prune" + / f"{tree.name}-{stamp}" + ) + try: + for rel in untracked: + src = tree / rel + if not src.exists() or src.is_symlink(): + continue + target = dest / rel + target.parent.mkdir(parents=True, exist_ok=True) + if src.is_dir(): + shutil.copytree(src, target, dirs_exist_ok=True) + else: + shutil.copy2(src, target) + return dest if dest.exists() else None + except Exception as exc: + logger.warning("Could not archive untracked files from %s: %s", tree, exc) + return None + + +def audit_worktrees(repo_root: str, *, with_sizes: bool = True) -> List[TreeRecord]: + """Classify every tree under ``.worktrees/`` without mutating anything.""" + import cli as _cli # lazy: cli.py is heavy + + worktrees_dir = Path(repo_root) / ".worktrees" + if not worktrees_dir.exists(): + return [] + + if _cli._repo_is_shallow(repo_root): + _cli._deepen_shallow_repo(repo_root) + + merge_cache = _cli._load_worktree_merge_cache() + cache_size_before = len(merge_cache) + + now = time.time() + records: List[TreeRecord] = [] + for entry in sorted(worktrees_dir.iterdir()): + if not entry.is_dir(): + continue + try: + age_days = (now - entry.stat().st_mtime) / 86400.0 + except Exception: + continue + size_mb = _tree_size_mb(entry) if with_sizes else None + + try: + branch_result = _git(["branch", "--show-current"], cwd=str(entry), timeout=5) + branch = branch_result.stdout.strip() + except Exception: + branch = "" + + def rec(verdict: str, reason: str, untracked: Optional[List[str]] = None): + records.append(TreeRecord( + name=entry.name, path=str(entry), branch=branch, + age_days=age_days, size_mb=size_mb, + verdict=verdict, reason=reason, + untracked=untracked or [], + )) + + if _KANBAN_RE.match(entry.name): + rec("keep", "kanban task tree (owned by kanban gc)") + continue + + lock_state = _cli._worktree_lock_is_live(repo_root, str(entry), timeout=5) + if lock_state == "live": + rec("keep", "in use by a running hermes session") + continue + + tracked_dirty, untracked = _dirty_split(str(entry)) + if tracked_dirty: + rec("keep", "uncommitted tracked changes (real work)") + continue + + if _cli._worktree_has_unpushed_commits(str(entry), timeout=5): + merged = _cli._worktree_commits_all_merged_upstream( + str(entry), timeout=30, cache=merge_cache, + max_ahead=_MAX_CHERRY_AHEAD, + ) + if not merged: + rec("keep", "unpushed commits not found upstream") + continue + + if untracked: + rec("reap-archive", + f"merged/pushed; {len(untracked)} untracked file(s) will be archived", + untracked) + else: + rec("reap", "clean and fully merged/pushed") + + if len(merge_cache) != cache_size_before: + _cli._save_worktree_merge_cache(merge_cache) + return records + + +def reclaim_worktrees( + repo_root: str, + *, + dry_run: bool = False, + records: Optional[List[TreeRecord]] = None, +) -> List[str]: + """Remove every reap-verdict tree from a frozen audit list. + + Operates ONLY on the provided (or freshly computed) audit records — never + re-globs inside the destructive loop, so trees created by concurrent + sessions after the audit are out of scope by construction. + """ + if records is None: + records = audit_worktrees(repo_root, with_sizes=False) + actions: List[str] = [] + for record in records: + if record.verdict not in {"reap", "reap-archive"}: + continue + if dry_run: + actions.append(f"would remove {record.name} ({record.reason})") + continue + + entry = Path(record.path) + if record.verdict == "reap-archive" and record.untracked: + archive = _archive_untracked(entry, record.untracked) + if archive is None: + actions.append(f"kept {record.name} (archive of untracked files failed)") + continue + actions.append(f"archived {len(record.untracked)} untracked file(s) → {archive}") + + # Dead-pid locks must be unlocked or `remove --force` refuses. + try: + _git(["worktree", "unlock", record.path], cwd=repo_root, timeout=10) + except Exception: + pass + + try: + remove_result = _git( + ["worktree", "remove", record.path, "--force"], + cwd=repo_root, timeout=30, + ) + if remove_result.returncode != 0: + actions.append( + f"failed to remove {record.name}: {remove_result.stderr.strip()}" + ) + continue + if record.branch and record.branch not in _PROTECTED_BRANCHES: + _git(["branch", "-D", record.branch], cwd=repo_root, timeout=10) + actions.append(f"removed {record.name}") + except Exception as exc: + actions.append(f"failed to remove {record.name}: {exc}") + + if not dry_run: + try: + _git(["worktree", "prune"], cwd=repo_root, timeout=15) + except Exception: + pass + return actions + + +def audit_branches(repo_root: str) -> List[BranchRecord]: + """Classify local branches: safe to delete when their content is on + upstream (fully merged OR every commit patch-equivalent via ``git + cherry``) and they are not checked out anywhere. + + Generalizes the startup pass's prefix list (``hermes/hermes-*``/``pr-*``) + to EVERY local branch, because deletion is gated on content reachability + rather than name: a branch whose commits are all upstream loses nothing + when its ref goes. Branch names checked out in any worktree, protected + names, and branches with unique commits are kept. + """ + import cli as _cli + + if _cli._repo_is_shallow(repo_root): + _cli._deepen_shallow_repo(repo_root) + + upstream = None + for candidate in ("origin/HEAD", "origin/main", "origin/master"): + probe = _git(["rev-parse", "--verify", "--quiet", candidate], cwd=repo_root, timeout=5) + if probe.returncode == 0: + upstream = candidate + break + if upstream is None: + return [] + + result = _git(["branch", "--format=%(refname:short)"], cwd=repo_root, timeout=10) + if result.returncode != 0: + return [] + branches = [b.strip() for b in result.stdout.splitlines() if b.strip()] + + active: set = set() + wt = _git(["worktree", "list", "--porcelain"], cwd=repo_root, timeout=10) + for line in wt.stdout.splitlines(): + if line.startswith("branch refs/heads/"): + active.add(line.split("branch refs/heads/", 1)[-1].strip()) + + merged_result = _git(["branch", "--merged", upstream, "--format=%(refname:short)"], + cwd=repo_root, timeout=15) + merged = {b.strip() for b in merged_result.stdout.splitlines() if b.strip()} + + def _classify_branch(branch: str) -> BranchRecord: + if branch in _PROTECTED_BRANCHES or branch in active: + return BranchRecord(branch, "keep", "protected or checked out") + if branch in merged: + return BranchRecord(branch, "delete", "fully merged into " + upstream) + # Rebase merges rewrite SHAs, so --merged misses them; cherry + # patch-equivalence catches the dominant leak. Bounded: a branch + # far ahead is a stale-base lane, keep it. + ahead = _git(["rev-list", "--count", f"{upstream}..{branch}"], cwd=repo_root, timeout=10) + try: + ahead_count = int(ahead.stdout.strip() or "0") + except ValueError: + ahead_count = _MAX_CHERRY_AHEAD + 1 + if ahead_count == 0: + return BranchRecord(branch, "delete", "no commits beyond " + upstream) + if ahead_count > _MAX_CHERRY_AHEAD: + return BranchRecord(branch, "keep", f"{ahead_count} commits ahead (stale-base lane)") + cherry = _git(["cherry", upstream, branch], cwd=repo_root, timeout=30) + if cherry.returncode != 0: + return BranchRecord(branch, "keep", "could not verify (git cherry failed)") + lines = [ln for ln in cherry.stdout.splitlines() if ln.strip()] + if lines and all(ln.startswith("-") for ln in lines): + return BranchRecord(branch, "delete", "all commits patch-equivalent upstream") + unique = sum(1 for ln in lines if ln.startswith("+")) + return BranchRecord(branch, "keep", f"{unique} unique commit(s) not upstream") + + # Read-only classification — parallel, like the tree audit (a busy + # multi-agent box carries hundreds of local branches; serial cherry + # probes at ~0.2-1s each make the audit minutes long). + import concurrent.futures + + workers = max(1, min(8, (os.cpu_count() or 4), len(branches))) + if workers > 1: + try: + with concurrent.futures.ThreadPoolExecutor( + max_workers=workers, thread_name_prefix="hermes-branch-gc" + ) as pool: + return list(pool.map(_classify_branch, branches)) + except Exception: + pass + return [_classify_branch(b) for b in branches] + + +def reclaim_branches( + repo_root: str, + *, + dry_run: bool = False, + records: Optional[List[BranchRecord]] = None, +) -> List[str]: + """Delete every delete-verdict branch from a frozen audit list.""" + if records is None: + records = audit_branches(repo_root) + actions: List[str] = [] + for record in records: + if record.verdict != "delete": + continue + if dry_run: + actions.append(f"would delete branch {record.name} ({record.reason})") + continue + result = _git(["branch", "-D", record.name], cwd=repo_root, timeout=10) + if result.returncode == 0: + actions.append(f"deleted branch {record.name}") + else: + actions.append(f"failed to delete {record.name}: {result.stderr.strip()}") + return actions + + +def worktrees_summary(repo_root: str) -> tuple[int, Optional[int]]: + """(tree_count, total_size_mb) for the escalation notice. Size is + best-effort with a hard timeout so the startup path never stalls.""" + worktrees_dir = Path(repo_root) / ".worktrees" + if not worktrees_dir.exists(): + return 0, None + try: + count = sum(1 for e in worktrees_dir.iterdir() if e.is_dir()) + except Exception: + return 0, None + size_mb: Optional[int] = None + try: + result = subprocess.run( + ["du", "-sm", str(worktrees_dir)], + capture_output=True, text=True, encoding="utf-8", + errors="replace", timeout=20, + ) + if result.returncode == 0 and result.stdout.strip(): + size_mb = int(result.stdout.split()[0]) + except Exception: + pass + return count, size_mb diff --git a/tests/hermes_cli/test_worktree_gc.py b/tests/hermes_cli/test_worktree_gc.py new file mode 100644 index 0000000000..03e24a51ed --- /dev/null +++ b/tests/hermes_cli/test_worktree_gc.py @@ -0,0 +1,243 @@ +"""Behavior contracts for hermes_cli.worktree_gc (attended reclaim). + +Each guard gets its own contract against a REAL git repo fixture (no mocks — +the entire value of these tests is exercising actual git verdicts): + +- clean + fully merged tree → reap +- untracked-only dirt → reap-archive (files archived, then removed) +- tracked modifications → keep, any age +- unique unpushed commits → keep +- patch-equivalent commits (rebase/squash-merge leak) → reap +- live-locked tree → keep +- kanban t_ tree → keep (owned by kanban gc) +- branch GC: merged branch deleted, unique-commit branch kept, + checked-out branch kept, protected names kept +- reclaim operates ONLY on the frozen audit list (concurrent-session trap) +""" + +import os +import subprocess +from pathlib import Path + +import pytest + +from hermes_cli import worktree_gc + + +def _git(args, cwd, env=None): + e = dict(os.environ) + e.update({ + "GIT_AUTHOR_NAME": "t", "GIT_AUTHOR_EMAIL": "t@t", + "GIT_COMMITTER_NAME": "t", "GIT_COMMITTER_EMAIL": "t@t", + }) + if env: + e.update(env) + result = subprocess.run( + ["git", *args], capture_output=True, text=True, cwd=str(cwd), env=e, + ) + assert result.returncode == 0, f"git {args} failed: {result.stderr}" + return result.stdout.strip() + + +@pytest.fixture +def repo(tmp_path, monkeypatch): + """origin (bare) + clone with .worktrees/, HOME redirected for archives.""" + monkeypatch.setenv("HOME", str(tmp_path / "home")) + (tmp_path / "home").mkdir() + + origin = tmp_path / "origin.git" + origin.mkdir() + _git(["init", "--bare", "-b", "main"], origin) + + clone = tmp_path / "repo" + _git(["clone", str(origin), str(clone)], tmp_path) + (clone / "README.md").write_text("hello\n") + _git(["add", "."], clone) + _git(["commit", "-m", "init"], clone) + _git(["push", "origin", "main"], clone) + # origin/HEAD so upstream resolution works like a real clone. + _git(["remote", "set-head", "origin", "main"], clone) + (clone / ".worktrees").mkdir() + return clone + + +def _add_worktree(repo_path, name, branch=None): + tree = repo_path / ".worktrees" / name + branch = branch or f"hermes/{name}" + _git(["worktree", "add", str(tree), "-b", branch], repo_path) + return tree, branch + + +def _verdict(records, name): + match = [record for record in records if record.name == name] + assert match, f"no record for {name}" + return match[0] + + +class TestAuditVerdicts: + def test_clean_merged_tree_reaps(self, repo): + _add_worktree(repo, "hermes-clean") + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + assert _verdict(records, "hermes-clean").verdict == "reap" + + def test_tracked_modifications_keep(self, repo): + tree, _ = _add_worktree(repo, "hermes-dirty") + (tree / "README.md").write_text("edited\n") + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + record = _verdict(records, "hermes-dirty") + assert record.verdict == "keep" + assert "tracked" in record.reason + + def test_untracked_only_is_reap_archive(self, repo): + tree, _ = _add_worktree(repo, "hermes-scratch") + (tree / "PR_BODY_DRAFT.md").write_text("draft\n") + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + record = _verdict(records, "hermes-scratch") + assert record.verdict == "reap-archive" + assert record.untracked == ["PR_BODY_DRAFT.md"] + + def test_unique_unpushed_commits_keep(self, repo): + tree, _ = _add_worktree(repo, "hermes-work") + (tree / "new.py").write_text("x = 1\n") + _git(["add", "."], tree) + _git(["commit", "-m", "unique work"], tree) + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + record = _verdict(records, "hermes-work") + assert record.verdict == "keep" + assert "unpushed" in record.reason + + def test_patch_equivalent_commits_reap(self, repo): + """The squash/rebase-merge leak: local commit unreachable from any + remote ref but patch-equivalent to an upstream commit → merged work.""" + tree, _ = _add_worktree(repo, "hermes-merged") + (tree / "feat.py").write_text("y = 2\n") + _git(["add", "."], tree) + _git(["commit", "-m", "feat"], tree) + sha = _git(["rev-parse", "HEAD"], tree) + # "Merge" it to main with a DIFFERENT committer so the cherry-pick + # produces a distinct sha (same-second identical-committer cherry + # picks can produce the identical sha — pitfall from the skill). + _git(["cherry-pick", sha], repo, + env={"GIT_COMMITTER_NAME": "other", "GIT_COMMITTER_EMAIL": "o@o"}) + _git(["push", "origin", "main"], repo) + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + assert _verdict(records, "hermes-merged").verdict == "reap" + + def test_live_locked_tree_keeps(self, repo): + tree, _ = _add_worktree(repo, "hermes-live") + _git(["worktree", "lock", str(tree), + "--reason", f"hermes pid={os.getpid()}"], repo) + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + record = _verdict(records, "hermes-live") + assert record.verdict == "keep" + assert "in use" in record.reason + + def test_kanban_tree_untouched(self, repo): + _add_worktree(repo, "t_deadbeef", branch="kanban/t_deadbeef") + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + record = _verdict(records, "t_deadbeef") + assert record.verdict == "keep" + assert "kanban" in record.reason + + +class TestReclaim: + def test_reap_removes_tree_and_branch(self, repo): + tree, branch = _add_worktree(repo, "hermes-clean") + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + actions = worktree_gc.reclaim_worktrees(str(repo), records=records) + assert any("removed hermes-clean" in a for a in actions) + assert not tree.exists() + probe = subprocess.run( + ["git", "rev-parse", "--verify", "--quiet", branch], + capture_output=True, text=True, cwd=str(repo), + ) + assert probe.returncode != 0, "branch should be gone with its tree" + + def test_untracked_files_archived_before_removal(self, repo): + tree, _ = _add_worktree(repo, "hermes-scratch") + (tree / "NOTES.md").write_text("important scribbles\n") + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + worktree_gc.reclaim_worktrees(str(repo), records=records) + assert not tree.exists() + archive_root = Path.home() / ".hermes" / "archive" / "worktree-prune" + archived = list(archive_root.rglob("NOTES.md")) + assert archived, "untracked file must be archived, not destroyed" + assert archived[0].read_text() == "important scribbles\n" + + def test_dry_run_changes_nothing(self, repo): + tree, _ = _add_worktree(repo, "hermes-clean") + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + actions = worktree_gc.reclaim_worktrees( + str(repo), dry_run=True, records=records + ) + assert any("would remove" in a for a in actions) + assert tree.exists() + + def test_frozen_list_ignores_trees_created_after_audit(self, repo): + """Concurrent-session trap: a tree created between audit and reclaim + must be out of scope by construction.""" + _add_worktree(repo, "hermes-old") + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + late_tree, _ = _add_worktree(repo, "hermes-late") + worktree_gc.reclaim_worktrees(str(repo), records=records) + assert late_tree.exists(), "tree created after the audit must survive" + + def test_dead_locked_tree_is_unlocked_and_reaped(self, repo): + tree, _ = _add_worktree(repo, "hermes-zombie") + _git(["worktree", "lock", str(tree), + "--reason", "hermes pid=999999999"], repo) + records = worktree_gc.audit_worktrees(str(repo), with_sizes=False) + assert _verdict(records, "hermes-zombie").verdict == "reap" + worktree_gc.reclaim_worktrees(str(repo), records=records) + assert not tree.exists() + + +class TestBranchGC: + def test_merged_branch_deleted_any_name(self, repo): + """Branch GC is content-gated, not name-gated: any fully-merged local + branch is safe to delete regardless of prefix.""" + _git(["branch", "salv-12345", "main"], repo) + _git(["branch", "feat/some-old-thing", "main"], repo) + records = worktree_gc.audit_branches(str(repo)) + by_name = {record.name: record for record in records} + assert by_name["salv-12345"].verdict == "delete" + assert by_name["feat/some-old-thing"].verdict == "delete" + worktree_gc.reclaim_branches(str(repo), records=records) + out = _git(["branch", "--format=%(refname:short)"], repo) + assert "salv-12345" not in out + assert "feat/some-old-thing" not in out + + def test_unique_commit_branch_kept(self, repo): + _git(["checkout", "-b", "feat/real-work"], repo) + (repo / "wip.py").write_text("z = 3\n") + _git(["add", "."], repo) + _git(["commit", "-m", "wip"], repo) + _git(["checkout", "main"], repo) + records = worktree_gc.audit_branches(str(repo)) + by_name = {record.name: record for record in records} + assert by_name["feat/real-work"].verdict == "keep" + assert "unique" in by_name["feat/real-work"].reason + + def test_patch_equivalent_branch_deleted(self, repo): + """Rebase-merged PR branch: SHAs differ from main but every commit is + patch-equivalent — the dominant branch leak.""" + _git(["checkout", "-b", "fix/landed"], repo) + (repo / "fix.py").write_text("a = 4\n") + _git(["add", "."], repo) + _git(["commit", "-m", "fix"], repo) + sha = _git(["rev-parse", "HEAD"], repo) + _git(["checkout", "main"], repo) + _git(["cherry-pick", sha], repo, + env={"GIT_COMMITTER_NAME": "other", "GIT_COMMITTER_EMAIL": "o@o"}) + _git(["push", "origin", "main"], repo) + records = worktree_gc.audit_branches(str(repo)) + by_name = {record.name: record for record in records} + assert by_name["fix/landed"].verdict == "delete" + assert "patch-equivalent" in by_name["fix/landed"].reason + + def test_checked_out_and_protected_kept(self, repo): + _tree, branch = _add_worktree(repo, "hermes-active") + records = worktree_gc.audit_branches(str(repo)) + by_name = {record.name: record for record in records} + assert by_name["main"].verdict == "keep" + assert by_name[branch].verdict == "keep" diff --git a/website/docs/user-guide/cli.md b/website/docs/user-guide/cli.md index 9c4e7a6aab..08a9401468 100644 --- a/website/docs/user-guide/cli.md +++ b/website/docs/user-guide/cli.md @@ -58,6 +58,43 @@ hermes -w # Interactive mode in worktree hermes -w -z "Fix issue #123" # Single query in worktree ``` +### Worktree cleanup + +`hermes -w` sessions create disposable worktrees under `/.worktrees/`. +A conservative pruner runs automatically at startup (it only removes clean, +fully-merged scratch trees past an age threshold), but preserved trees and +merged local branches still accumulate on busy machines. Reclaim them +explicitly: + +```bash +hermes worktree list # audit: age, size, verdict, reason per tree +hermes worktree prune # remove safe trees + delete merged branches +hermes worktree prune --dry-run # show the plan without changing anything +hermes worktree prune --trees-only # leave local branches alone +hermes worktree prune --branches-only # leave worktrees alone +``` + +Inside a session, `/worktree prune [--dry-run]` does the same (and never +touches the tree the session is running in). + +Safety guarantees (all modes, any age): + +- Uncommitted **tracked** changes are never deleted. +- **Unique unpushed commits** are never deleted — commits that were + rebase/squash-merged upstream are detected via `git cherry` + patch-equivalence and count as merged, which is what lets the dominant + "merged PR, tree preserved forever" leak finally reclaim. +- Trees **in use by a running hermes session** are never touched. +- **Untracked-only scratch** (PR body drafts, notes) is archived to + `~/.hermes/archive/worktree-prune/` before its tree is removed — never + destroyed. +- Branch deletion is content-gated, not name-gated: any local branch whose + commits are all on upstream is safe to delete; branches with unique work, + checked-out branches, and `main`/`master`/`develop` are always kept. + +When `.worktrees/` grows past 10 trees or 5 GB, startup prints a one-line +notice pointing at these commands. + ### Plugin management The `hermes plugins` commands manage native Hermes plugins and portable Agent From fc9b4186a044b314d3d76ab47f4d123e3dd12a0b Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 23:39:40 -0700 Subject: [PATCH 34/42] fix(worktree): pruner reaps rebase-merged trees via PR state; worktree add survives disk contention MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two gaps behind the recurring 'hermes -w timed out after 30 seconds': 1. Rebase-merge leak: git cherry only catches patch-identical commits. Salvage flows routinely change the diff (conflict resolution, follow-up commits), so 12 of 22 'unpushed' trees on the incident box had MERGED PRs and were preserved forever. The pruner now falls back to 'gh pr list --head --state merged' — authoritative, memoized on (branch, head_sha) with True-only caching, fail-safe to preserve. 2. Creation timeout 30s -> 120s: the ~10k-file checkout measured 113s at near-zero CPU under multi-agent disk contention vs 1.2s idle. 30s killed legitimate creates and threw away completed work. --- cli.py | 80 ++++++++++++++++++++++++++++- tests/cli/test_worktree.py | 102 +++++++++++++++++++++++++++++++++++++ 2 files changed, 180 insertions(+), 2 deletions(-) diff --git a/cli.py b/cli.py index e026a56138..ffa623875b 100644 --- a/cli.py +++ b/cli.py @@ -1866,9 +1866,14 @@ def _setup_worktree(repo_root: str = None, sync_base: bool = True, "-c", "checkout.thresholdForParallelism=100", ] try: + # 120s, not 30: on a multi-agent box the ~10k-file materialization + # contends with sibling sessions' checkouts/fetches/Electron dev + # builds for the same disk — measured 113s wall at near-zero CPU + # under load vs 1.2s idle (Aug 2026). A too-tight timeout kills a + # legitimately slow create and wastes the work already done. result = subprocess.run( ["git", *_wt_add_cfg, "worktree", "add", str(wt_path), "-b", branch_name, base_ref], - capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, cwd=repo_root, + capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=120, cwd=repo_root, ) if result.returncode != 0: # If branching from the resolved remote ref failed for any reason @@ -1883,7 +1888,7 @@ def _setup_worktree(repo_root: str = None, sync_base: bool = True, base_ref, base_label = "HEAD", "HEAD (fallback — remote base failed)" result = subprocess.run( ["git", "worktree", "add", str(wt_path), "-b", branch_name, base_ref], - capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=30, cwd=repo_root, + capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=120, cwd=repo_root, ) if result.returncode != 0: _cleanup_failed_worktree_add(repo_root, wt_path, branch_name) @@ -2292,6 +2297,69 @@ def _worktree_commits_all_merged_upstream( return False +def _worktree_branch_pr_merged( + worktree_path: str, + timeout: int = 15, + cache: Optional[Dict[str, bool]] = None, +) -> bool: + """Return whether the worktree branch's PR is MERGED on GitHub. + + Escape hatch for the case ``git cherry`` cannot catch: a rebase-merge that + altered the diff (conflict resolution against a moved base, follow-up + commits added during salvage/CI-fix) changes the patch-id, so the local + commits are no longer patch-equivalent to anything upstream even though + the PR merged. Those trees survive the cherry check forever (Aug 2026: + 12 of 22 "unpushed" trees on a loaded box had MERGED PRs). + + GitHub's PR state is the authoritative merge signal, so a clean tree + whose branch has a MERGED PR is reaped. Verdicts are memoized keyed on + ``(branch, head_sha)`` — MERGED is monotonic, so a True verdict is cached + permanently; False is never cached (the PR may merge later without new + local commits, which would leave the key unchanged). + + Fails SAFE toward False (preserve): no gh binary, offline, rate-limited, + detached HEAD, or any parse failure keeps the tree. + """ + import subprocess + + try: + head = subprocess.run( + ["git", "rev-parse", "--abbrev-ref", "HEAD"], + capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path, + ) + if head.returncode != 0: + return False + branch = head.stdout.strip() + if not branch or branch == "HEAD": # detached — no PR to look up + return False + + cache_key = None + if cache is not None: + sha = subprocess.run( + ["git", "rev-parse", "HEAD"], + capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path, + ) + if sha.returncode == 0 and sha.stdout.strip(): + cache_key = f"pr-merged:{branch}:{sha.stdout.strip()}" + if cache.get(cache_key) is True: + return True + + result = subprocess.run( + ["gh", "pr", "list", "--head", branch, "--state", "merged", + "--json", "number", "--limit", "1"], + capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout, cwd=worktree_path, + ) + if result.returncode != 0: + return False + prs = json.loads(result.stdout or "[]") + merged = isinstance(prs, list) and len(prs) > 0 + if merged and cache is not None and cache_key is not None: + cache[cache_key] = True + return merged + except Exception: + return False + + def _worktree_lock_is_live(repo_root: str, worktree_path: str, timeout: int = 10): """Classify a worktree's git lock as live, dead, or absent. @@ -2652,6 +2720,14 @@ def _prune_stale_worktrees(repo_root: str, max_age_hours: int = 24) -> None: merged = _worktree_commits_all_merged_upstream( str(entry), timeout=30, cache=snapshot ) + if not merged: + # Rebase-merge escape hatch: conflict resolution or follow-up + # commits change the patch-id, so cherry misses them — but + # GitHub knows the PR merged. Authoritative and cheap (~0.3s, + # memoized on (branch, head_sha) so it's paid once per tree). + merged = _worktree_branch_pr_merged( + str(entry), timeout=15, cache=snapshot + ) with cache_lock: merge_cache.update(snapshot) if not merged: diff --git a/tests/cli/test_worktree.py b/tests/cli/test_worktree.py index aab76a6193..c2fe4119c0 100644 --- a/tests/cli/test_worktree.py +++ b/tests/cli/test_worktree.py @@ -1340,3 +1340,105 @@ class TestShallowCloneDeepening: assert wt.exists(), ( "genuinely unpushed commit must survive even after deepening" ) + + +class TestPrMergedEscapeHatch: + """Rebase-merged PRs whose diff changed during salvage defeat ``git + cherry`` (patch-id mismatch), so the pruner asks GitHub whether the + branch's PR is MERGED. These tests stub the ``gh`` binary on PATH — the + contract is about how the pruner consumes the answer, not about GitHub. + + Contract: + - gh reports a merged PR + tree is clean -> reaped + - gh reports no merged PR -> preserved + - gh missing/failing -> preserved (fail safe) + - dirty tree -> never reaped regardless of gh + """ + + _age = staticmethod(TestWorktreeLockReaping._age) + + @staticmethod + def _mk_diverged(repo, name, age_h=100): + """Worktree with a commit NOT patch-equivalent to anything upstream.""" + p = repo / ".worktrees" / name + (repo / ".worktrees").mkdir(exist_ok=True) + subprocess.run( + ["git", "worktree", "add", str(p), "-b", f"hermes/{name}", "HEAD"], + cwd=repo, capture_output=True, + ) + (p / "salvaged.txt").write_text("diff that was reworked during salvage\n") + subprocess.run(["git", "add", "salvaged.txt"], cwd=p, capture_output=True) + subprocess.run(["git", "commit", "-m", "salvaged work"], cwd=p, capture_output=True) + TestPrMergedEscapeHatch._age(p, age_h) + return p + + @staticmethod + def _stub_gh(tmp_path, monkeypatch, stdout='[{"number": 1}]', exit_code=0): + gh = tmp_path / "bin" / "gh" + gh.parent.mkdir(parents=True, exist_ok=True) + gh.write_text(f"#!/bin/sh\nprintf '%s' '{stdout}'\nexit {exit_code}\n") + gh.chmod(0o755) + monkeypatch.setenv("PATH", f"{gh.parent}:{os.environ['PATH']}") + + def test_merged_pr_tree_is_reaped(self, git_repo, tmp_path, monkeypatch): + import cli + wt = self._mk_diverged(git_repo, "hermes-rebase-merged") + assert cli._worktree_commits_all_merged_upstream(str(wt)) is False, ( + "precondition: cherry must NOT consider this merged — the PR " + "check is the only thing that can reap it" + ) + self._stub_gh(tmp_path, monkeypatch) + cli._prune_stale_worktrees(str(git_repo)) + assert not wt.exists(), ( + "clean tree whose branch has a MERGED PR is merged work — reap it" + ) + + def test_no_merged_pr_preserved(self, git_repo, tmp_path, monkeypatch): + import cli + wt = self._mk_diverged(git_repo, "hermes-pr-open") + self._stub_gh(tmp_path, monkeypatch, stdout="[]") + cli._prune_stale_worktrees(str(git_repo)) + assert wt.exists(), "no merged PR -> still unpushed work, preserve" + + def test_gh_failure_fails_safe(self, git_repo, tmp_path, monkeypatch): + import cli + wt = self._mk_diverged(git_repo, "hermes-gh-down") + self._stub_gh(tmp_path, monkeypatch, stdout="", exit_code=1) + cli._prune_stale_worktrees(str(git_repo)) + assert wt.exists(), "gh failure must preserve the tree (fail safe)" + + def test_dirty_tree_never_reaped_even_with_merged_pr( + self, git_repo, tmp_path, monkeypatch + ): + import cli + wt = self._mk_diverged(git_repo, "hermes-dirty-merged") + (wt / "uncommitted.txt").write_text("in-flight\n") + self._age(wt, 100) + self._stub_gh(tmp_path, monkeypatch) + cli._prune_stale_worktrees(str(git_repo)) + assert wt.exists(), "dirty guard outranks the PR-merged verdict" + + def test_merged_verdict_memoized_by_branch_and_head( + self, git_repo, tmp_path, monkeypatch + ): + import cli + wt = self._mk_diverged(git_repo, "hermes-memo") + self._stub_gh(tmp_path, monkeypatch) + cache: dict = {} + assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is True + keys = [k for k in cache if k.startswith("pr-merged:")] + assert len(keys) == 1 and cache[keys[0]] is True + # Break gh: a cached True verdict must not re-consult it. + self._stub_gh(tmp_path, monkeypatch, stdout="", exit_code=1) + assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is True + + def test_negative_verdict_not_cached(self, git_repo, tmp_path, monkeypatch): + import cli + wt = self._mk_diverged(git_repo, "hermes-nocache-neg") + self._stub_gh(tmp_path, monkeypatch, stdout="[]") + cache: dict = {} + assert cli._worktree_branch_pr_merged(str(wt), cache=cache) is False + assert not [k for k in cache if k.startswith("pr-merged:")], ( + "False must not be memoized — the PR can merge later with the " + "same (branch, head) key" + ) From db5d5dffeae128a7809f176c820ccc81bfdc52dc Mon Sep 17 00:00:00 2001 From: joaomarcos Date: Tue, 18 Aug 2026 17:18:21 -0300 Subject: [PATCH 35/42] fix(agent): guard against uncompressed session overflow when compression is disabled (#89297) When compression is explicitly disabled (compression.enabled: false), conversations can grow past the model's context window across hundreds of messages (e.g., 824 messages / 460K+ tokens in #89297). Serializing massive JSON payloads repeatedly under memory-constrained environments leads to swap thrashing (STAT=U) and unhandled provider errors. Add a pre-flight uncompressed context overflow guardrail in build_turn_context and a deduped _warn_uncompressed_context_overflow method on AIAgent to alert users to run /compact or enable compression before unmanageable payloads freeze the process. --- agent/turn_context.py | 43 +++++++++++ run_agent.py | 20 ++++++ .../test_uncompressed_context_guardrail.py | 71 +++++++++++++++++++ 3 files changed, 134 insertions(+) create mode 100644 tests/agent/test_uncompressed_context_guardrail.py diff --git a/agent/turn_context.py b/agent/turn_context.py index 650cfe7679..9128c75e5f 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -1163,6 +1163,49 @@ def build_turn_context( agent._last_content_with_tools = None agent._last_content_tools_all_housekeeping = False agent._mute_post_response = False + elif not agent.compression_enabled: + # Uncompressed session guard (#89297): when compression is explicitly + # disabled, sessions can grow past the model's context window across + # hundreds of messages without compression to shrink them. + # Run a cheap character pre-check before computing rough tokens. + _raw_chars = sum( + len(m.get("content") or "") for m in messages if isinstance(m, dict) + ) + if _raw_chars > 20_000: + _uncompressed_tokens = estimate_request_tokens_rough( + messages, + system_prompt=active_system_prompt or "", + tools=agent.tools or None, + ) + _ctx_len = getattr( + getattr(agent, "context_compressor", None), "context_length", None + ) + if not isinstance(_ctx_len, int) or _ctx_len <= 0: + try: + from agent.model_metadata import get_model_context_length + + _ctx_len = get_model_context_length( + agent.model, + getattr(agent, "base_url", "") or "", + provider=getattr(agent, "provider", "") or "", + ) + except Exception: + _ctx_len = None + if _ctx_len and _uncompressed_tokens > _ctx_len: + _warn_fn = getattr( + agent, "_warn_uncompressed_context_overflow", None + ) + if callable(_warn_fn): + _warn_fn(_uncompressed_tokens, _ctx_len) + else: + _emit_w = getattr(agent, "_emit_warning", None) + if callable(_emit_w): + _emit_w( + f"⚠️ Session context (~{_uncompressed_tokens:,} tokens) exceeds the " + f"model context window (~{_ctx_len:,} tokens) with compression disabled " + f"(compression.enabled: false). Use /compact to compress history or " + f"enable compression in config.yaml." + ) if _preflight_compressed: # Compression rebuilt the list (tail messages are fresh compaction diff --git a/run_agent.py b/run_agent.py index 007726c463..ac06aafaf1 100644 --- a/run_agent.py +++ b/run_agent.py @@ -1042,6 +1042,26 @@ class AIAgent: ) ) + def _warn_uncompressed_context_overflow( + self, preflight_tokens: int, context_length: int + ) -> None: + """Surface a deduped warning when uncompressed context exceeds model limit. + + When compression is explicitly disabled (compression.enabled: false), long + sessions can grow past the model context window with no compression to shrink + them (#89297). Surface an actionable warning so the user knows to run /compact + or enable compression. + """ + _warn_key = ("uncompressed_ctx_overflow", context_length) + if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key: + self._last_ctx_overflow_warn = _warn_key + self._emit_warning( + f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model " + f"context window (~{context_length:,} tokens) with compression disabled " + f"(compression.enabled: false). Use /compact to compress history or " + f"enable compression in config.yaml." + ) + def _clear_context_overflow_warn(self) -> None: """Reset the dedup state for the blocked-overflow warning. diff --git a/tests/agent/test_uncompressed_context_guardrail.py b/tests/agent/test_uncompressed_context_guardrail.py new file mode 100644 index 0000000000..c509c13460 --- /dev/null +++ b/tests/agent/test_uncompressed_context_guardrail.py @@ -0,0 +1,71 @@ +"""Unit tests for uncompressed context overflow guardrail (Issue #89297).""" + +from __future__ import annotations + +import types +from unittest.mock import MagicMock, patch + +import pytest + +from agent.turn_context import TurnContext, build_turn_context +from tests.agent.test_turn_context import _FakeAgent, _build + + +class _FakeUncompressedAgent(_FakeAgent): + """Agent stub with compression disabled (compression.enabled: False).""" + + def __init__(self, model="deepseek-v4-flash", context_length=10_000): + super().__init__() + self.model = model + self.provider = "deepseek" + self.compression_enabled = False + self.context_compressor = types.SimpleNamespace( + protect_first_n=2, + protect_last_n=2, + context_length=context_length, + threshold_tokens=int(context_length * 0.75), + last_prompt_tokens=-1, + ) + + def _warn_uncompressed_context_overflow(self, preflight_tokens: int, context_length: int) -> None: + _warn_key = ("uncompressed_ctx_overflow", context_length) + if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key: + self._last_ctx_overflow_warn = _warn_key + msg = ( + f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model " + f"context window (~{context_length:,} tokens) with compression disabled " + f"(compression.enabled: false). Use /compact to compress history or " + f"enable compression in config.yaml." + ) + self._emit_warning(msg) + + +def test_uncompressed_session_within_limits_emits_no_warning(): + agent = _FakeUncompressedAgent(context_length=128_000) + history = [ + {"role": "user", "content": "hello"}, + {"role": "assistant", "content": "hi there"}, + ] + tctx = _build(agent, conversation_history=history) + assert isinstance(tctx, TurnContext) + agent._emit_warning.assert_not_called() + + +def test_uncompressed_session_exceeding_context_limit_warns(): + # Model context length is 10,000 tokens (~40,000 chars) + agent = _FakeUncompressedAgent(context_length=10_000) + + # Construct an oversized uncompressed history of ~15,000 tokens (>60,000 chars) + large_turn = "Large context content " * 500 # ~2,500 tokens + history = [] + for i in range(10): + history.append({"role": "user", "content": f"Turn {i}: {large_turn}"}) + history.append({"role": "assistant", "content": f"Reply {i}: {large_turn}"}) + + tctx = _build(agent, conversation_history=history) + assert isinstance(tctx, TurnContext) + agent._emit_warning.assert_called_once() + warning_msg = agent._emit_warning.call_args[0][0] + assert "exceeds the model context window" in warning_msg + assert "compression.enabled: false" in warning_msg + assert "10,000 tokens" in warning_msg From 4d1fc6ca0acb01236f520c1b0e75dcfa673b0abd Mon Sep 17 00:00:00 2001 From: joaomarcos Date: Tue, 18 Aug 2026 17:30:29 -0300 Subject: [PATCH 36/42] fix(agent): add mid-turn uncompressed context overflow guardrail (#89297) --- agent/conversation_loop.py | 26 +++++++++++++++++++++++++- 1 file changed, 25 insertions(+), 1 deletion(-) diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 768d8e019e..44f78ac5e6 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -2765,7 +2765,31 @@ def run_conversation( request_pressure_tokens, int(getattr(_compressor, "threshold_tokens", 0) or 0), ) - + elif not agent.compression_enabled and len(messages) > 1: + # Uncompressed session guard (#89297) - Defense in Depth: + # If mid-turn tool results expand the context beyond model limits + # while compression is disabled, warn the operator in-loop. + _ctx_len = getattr( + getattr(agent, "context_compressor", None), "context_length", None + ) + if not isinstance(_ctx_len, int) or _ctx_len <= 0: + try: + from agent.model_metadata import get_model_context_length + + _ctx_len = get_model_context_length( + agent.model, + getattr(agent, "base_url", "") or "", + provider=getattr(agent, "provider", "") or "", + ) + except Exception: + _ctx_len = None + if _ctx_len and request_pressure_tokens > _ctx_len: + _warn_fn = getattr( + agent, "_warn_uncompressed_context_overflow", None + ) + if callable(_warn_fn): + _warn_fn(request_pressure_tokens, _ctx_len) + # Thinking spinner for quiet mode (animated during API call) thinking_spinner = None From cef999c5cffb1fcbe4799fe159e041ab6016d1e2 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Thu, 20 Aug 2026 12:06:27 +0530 Subject: [PATCH 37/42] refactor(agent): consolidate uncompressed-overflow guard to one warn site + re-arm MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-up on the salvaged #89444: - Warn fires only from the conversation-loop pre-API site, reusing the unconditionally computed request_pressure_tokens (zero marginal cost, covers turn-start AND mid-turn growth) — drops the duplicate every-turn estimate the turn-context block paid. - Turn-context block now only RE-ARMS the dedup once the session is back under the window, so warn -> /compress -> regrow warns again (the dedup was previously never cleared with compression disabled). - Char pre-check treats non-string (multimodal) content as over-gate — len() of a part list defeated the 20k char floor (probe: 10 'chars' vs ~70k real tokens) — and compares against the window, not a flat 20k. - Deletes the unreachable get_model_context_length fallback from both sites (context_compressor always exists; its context_length property hard-floors positive; the fallback would have been a synchronous network probe mid-turn that also bypassed config overrides) and the undeduped inline _emit_warning fallback (third copy of the message). - Tests bind the PRODUCTION warn/clear methods (previously a verbatim fake reimplementation left them uncovered) and add dedup, re-arm, no-rearm-while-over, and multimodal-gate coverage. --- agent/conversation_loop.py | 33 ++-- agent/turn_context.py | 84 +++++---- .../test_uncompressed_context_guardrail.py | 174 ++++++++++++++---- 3 files changed, 207 insertions(+), 84 deletions(-) diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 44f78ac5e6..951b874701 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -2766,24 +2766,27 @@ def run_conversation( int(getattr(_compressor, "threshold_tokens", 0) or 0), ) elif not agent.compression_enabled and len(messages) > 1: - # Uncompressed session guard (#89297) - Defense in Depth: - # If mid-turn tool results expand the context beyond model limits - # while compression is disabled, warn the operator in-loop. + # Uncompressed session guard (#89297): compression is disabled, so + # nothing shrinks a growing session. Reuse the unconditionally + # computed request estimate (zero marginal cost — this site runs + # before every provider request, covering turn-start AND mid-turn + # tool-result growth) and surface a deduped, actionable warning + # when the request exceeds the model context window. The dedup is + # re-armed by the turn-context preflight once the session is back + # under the window (manual /compress works with compression + # disabled), so the guard warns again on a later re-overflow. + # context_compressor always exists (agent_init constructs it even + # when compression is disabled) and its context_length property + # hard-floors at a positive default — no metadata re-resolution + # needed here. _ctx_len = getattr( getattr(agent, "context_compressor", None), "context_length", None ) - if not isinstance(_ctx_len, int) or _ctx_len <= 0: - try: - from agent.model_metadata import get_model_context_length - - _ctx_len = get_model_context_length( - agent.model, - getattr(agent, "base_url", "") or "", - provider=getattr(agent, "provider", "") or "", - ) - except Exception: - _ctx_len = None - if _ctx_len and request_pressure_tokens > _ctx_len: + if ( + isinstance(_ctx_len, int) + and _ctx_len > 0 + and request_pressure_tokens > _ctx_len + ): _warn_fn = getattr( agent, "_warn_uncompressed_context_overflow", None ) diff --git a/agent/turn_context.py b/agent/turn_context.py index 9128c75e5f..df8969a666 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -1166,46 +1166,54 @@ def build_turn_context( elif not agent.compression_enabled: # Uncompressed session guard (#89297): when compression is explicitly # disabled, sessions can grow past the model's context window across - # hundreds of messages without compression to shrink them. - # Run a cheap character pre-check before computing rough tokens. - _raw_chars = sum( - len(m.get("content") or "") for m in messages if isinstance(m, dict) + # hundreds of messages with nothing to shrink them. The warning itself + # fires from the conversation loop's pre-API site, which reuses the + # unconditionally computed request estimate at zero marginal cost and + # covers both turn-start and mid-turn growth (every provider request + # passes through it). Here we only RE-ARM the dedup once the session + # is back under the window, so the guard can warn again after the + # user compacts (/compress with force=True works with compression + # disabled) and the context later regrows past the limit. + _ctx_len = getattr( + getattr(agent, "context_compressor", None), "context_length", None ) - if _raw_chars > 20_000: - _uncompressed_tokens = estimate_request_tokens_rough( - messages, - system_prompt=active_system_prompt or "", - tools=agent.tools or None, - ) - _ctx_len = getattr( - getattr(agent, "context_compressor", None), "context_length", None - ) - if not isinstance(_ctx_len, int) or _ctx_len <= 0: - try: - from agent.model_metadata import get_model_context_length - - _ctx_len = get_model_context_length( - agent.model, - getattr(agent, "base_url", "") or "", - provider=getattr(agent, "provider", "") or "", - ) - except Exception: - _ctx_len = None - if _ctx_len and _uncompressed_tokens > _ctx_len: - _warn_fn = getattr( - agent, "_warn_uncompressed_context_overflow", None + if isinstance(_ctx_len, int) and _ctx_len > 0: + _raw_chars = 0 + for _m in messages: + if not isinstance(_m, dict): + continue + _c = _m.get("content") + if isinstance(_c, str): + _raw_chars += len(_c) + elif _c: + # Non-string, non-empty content (multimodal part lists, + # dict payloads) defeats a char count — force the real + # estimate by treating it as over-gate. None/"" (routine + # assistant tool-call rows) contribute nothing. + _raw_chars = _ctx_len + 1 + break + # Cheap gate: a session whose raw text is under ~1/4 of the + # window (4 chars/token upper bound) cannot be over it — skip + # the estimator. Non-string (multimodal) content defeats a char + # count, so any such message forces the real estimate. + if _raw_chars <= _ctx_len: + _clear_warn = getattr( + agent, "_clear_context_overflow_warn", None ) - if callable(_warn_fn): - _warn_fn(_uncompressed_tokens, _ctx_len) - else: - _emit_w = getattr(agent, "_emit_warning", None) - if callable(_emit_w): - _emit_w( - f"⚠️ Session context (~{_uncompressed_tokens:,} tokens) exceeds the " - f"model context window (~{_ctx_len:,} tokens) with compression disabled " - f"(compression.enabled: false). Use /compact to compress history or " - f"enable compression in config.yaml." - ) + if callable(_clear_warn): + _clear_warn() + else: + _uncompressed_tokens = estimate_request_tokens_rough( + messages, + system_prompt=active_system_prompt or "", + tools=agent.tools or None, + ) + if _uncompressed_tokens <= _ctx_len: + _clear_warn = getattr( + agent, "_clear_context_overflow_warn", None + ) + if callable(_clear_warn): + _clear_warn() if _preflight_compressed: # Compression rebuilt the list (tail messages are fresh compaction diff --git a/tests/agent/test_uncompressed_context_guardrail.py b/tests/agent/test_uncompressed_context_guardrail.py index c509c13460..04a912d0bb 100644 --- a/tests/agent/test_uncompressed_context_guardrail.py +++ b/tests/agent/test_uncompressed_context_guardrail.py @@ -1,18 +1,35 @@ -"""Unit tests for uncompressed context overflow guardrail (Issue #89297).""" +"""Uncompressed context overflow guardrail (#89297). + +When compression is explicitly disabled (``compression.enabled: false``), +sessions can grow past the model context window with nothing to shrink them. +The conversation loop's pre-API site warns (deduped, actionable); the +turn-context preflight re-arms the dedup once the session is back under the +window so a later re-overflow warns again. + +The fake binds the PRODUCTION ``_warn_uncompressed_context_overflow`` / +``_clear_context_overflow_warn`` methods so their dedup logic is actually +under test (not a reimplementation). +""" from __future__ import annotations import types -from unittest.mock import MagicMock, patch +from unittest.mock import MagicMock -import pytest - -from agent.turn_context import TurnContext, build_turn_context +from agent.turn_context import TurnContext, build_turn_context # noqa: F401 +from run_agent import AIAgent from tests.agent.test_turn_context import _FakeAgent, _build class _FakeUncompressedAgent(_FakeAgent): - """Agent stub with compression disabled (compression.enabled: False).""" + """Agent stub with compression disabled, bound to the REAL warn methods.""" + + # Production methods under test — bound from AIAgent so the dedup key + # handling and message text cannot silently drift from what ships. + _warn_uncompressed_context_overflow = ( + AIAgent._warn_uncompressed_context_overflow + ) + _clear_context_overflow_warn = AIAgent._clear_context_overflow_warn def __init__(self, model="deepseek-v4-flash", context_length=10_000): super().__init__() @@ -27,21 +44,47 @@ class _FakeUncompressedAgent(_FakeAgent): last_prompt_tokens=-1, ) - def _warn_uncompressed_context_overflow(self, preflight_tokens: int, context_length: int) -> None: - _warn_key = ("uncompressed_ctx_overflow", context_length) - if getattr(self, "_last_ctx_overflow_warn", None) != _warn_key: - self._last_ctx_overflow_warn = _warn_key - msg = ( - f"⚠️ Session context (~{preflight_tokens:,} tokens) exceeds the model " - f"context window (~{context_length:,} tokens) with compression disabled " - f"(compression.enabled: false). Use /compact to compress history or " - f"enable compression in config.yaml." - ) - self._emit_warning(msg) + +def _oversized_history(n_turns: int = 10) -> list: + large_turn = "Large context content " * 500 # ~2,500 tokens each + history = [] + for i in range(n_turns): + history.append({"role": "user", "content": f"Turn {i}: {large_turn}"}) + history.append({"role": "assistant", "content": f"Reply {i}: {large_turn}"}) + return history + + +def test_production_warn_emits_once_and_dedups(): + """The real method warns once, then dedups identical overflows.""" + agent = _FakeUncompressedAgent(context_length=10_000) + agent._emit_warning = MagicMock() + + agent._warn_uncompressed_context_overflow(15_000, 10_000) + agent._warn_uncompressed_context_overflow(16_000, 10_000) + + agent._emit_warning.assert_called_once() + msg = agent._emit_warning.call_args[0][0] + assert "exceeds the model context window" in msg + assert "compression.enabled: false" in msg + assert "10,000 tokens" in msg + + +def test_clear_rearms_the_warning(): + """After _clear_context_overflow_warn (session back under the window), + a later re-overflow warns again.""" + agent = _FakeUncompressedAgent(context_length=10_000) + agent._emit_warning = MagicMock() + + agent._warn_uncompressed_context_overflow(15_000, 10_000) + agent._clear_context_overflow_warn() + agent._warn_uncompressed_context_overflow(15_500, 10_000) + + assert agent._emit_warning.call_count == 2 def test_uncompressed_session_within_limits_emits_no_warning(): agent = _FakeUncompressedAgent(context_length=128_000) + agent._emit_warning = MagicMock() history = [ {"role": "user", "content": "hello"}, {"role": "assistant", "content": "hi there"}, @@ -51,21 +94,90 @@ def test_uncompressed_session_within_limits_emits_no_warning(): agent._emit_warning.assert_not_called() -def test_uncompressed_session_exceeding_context_limit_warns(): - # Model context length is 10,000 tokens (~40,000 chars) - agent = _FakeUncompressedAgent(context_length=10_000) - - # Construct an oversized uncompressed history of ~15,000 tokens (>60,000 chars) - large_turn = "Large context content " * 500 # ~2,500 tokens - history = [] - for i in range(10): - history.append({"role": "user", "content": f"Turn {i}: {large_turn}"}) - history.append({"role": "assistant", "content": f"Reply {i}: {large_turn}"}) +def test_preflight_rearm_clears_dedup_when_back_under_window(): + """The turn-context preflight re-arms the dedup once the session fits + again (e.g. after a manual /compress), so growth past the window later + warns a second time.""" + agent = _FakeUncompressedAgent(context_length=128_000) + agent._emit_warning = MagicMock() + # Simulate a previously fired warning. + agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000) + history = [ + {"role": "user", "content": "hello"}, + {"role": "assistant", "content": "hi there"}, + ] tctx = _build(agent, conversation_history=history) assert isinstance(tctx, TurnContext) + + # Preflight cleared the dedup — the next overflow warns again. + assert agent._last_ctx_overflow_warn is None + agent._warn_uncompressed_context_overflow(200_000, 128_000) agent._emit_warning.assert_called_once() - warning_msg = agent._emit_warning.call_args[0][0] - assert "exceeds the model context window" in warning_msg - assert "compression.enabled: false" in warning_msg - assert "10,000 tokens" in warning_msg + + +def test_preflight_does_not_rearm_while_still_over_window(): + """While the session is still over the window, the dedup must survive + the preflight (no per-turn warn spam).""" + agent = _FakeUncompressedAgent(context_length=10_000) + agent._emit_warning = MagicMock() + agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 10_000) + + tctx = _build(agent, conversation_history=_oversized_history()) + assert isinstance(tctx, TurnContext) + + assert agent._last_ctx_overflow_warn == ("uncompressed_ctx_overflow", 10_000) + agent._emit_warning.assert_not_called() + + +def test_multimodal_content_forces_real_estimate_in_rearm_gate(): + """List (multimodal) content defeats a char count; the pre-check must + treat it as over-gate so the real estimator decides. A tiny multimodal + session is still under the window, so the dedup is re-armed.""" + agent = _FakeUncompressedAgent(context_length=128_000) + agent._emit_warning = MagicMock() + agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000) + + history = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "look at this"}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAAA"}}, + ], + }, + {"role": "assistant", "content": "looking"}, + ] + tctx = _build(agent, conversation_history=history) + assert isinstance(tctx, TurnContext) + assert agent._last_ctx_overflow_warn is None + + +def test_none_content_tool_call_rows_do_not_defeat_cheap_gate(): + """Assistant tool-call rows routinely carry content=None; they must + count as zero chars (NOT force the estimator) so the cheap gate keeps + its value in ordinary tool-using sessions. Regression for the salvage + follow-up's first draft, where `None` hit the over-gate branch.""" + from unittest.mock import patch as _patch + + agent = _FakeUncompressedAgent(context_length=128_000) + agent._emit_warning = MagicMock() + agent._last_ctx_overflow_warn = ("uncompressed_ctx_overflow", 128_000) + + history = [ + {"role": "user", "content": "run the tool"}, + {"role": "assistant", "content": None, + "tool_calls": [{"id": "c1", "function": {"name": "t", "arguments": "{}"}}]}, + {"role": "tool", "tool_call_id": "c1", "content": "small result"}, + {"role": "assistant", "content": "done"}, + ] + with _patch( + "agent.turn_context.estimate_request_tokens_rough" + ) as mock_est: + tctx = _build(agent, conversation_history=history) + + assert isinstance(tctx, TurnContext) + # Cheap gate decided (tiny session, under window): estimator never ran, + # and the dedup was still re-armed via the raw-chars branch. + mock_est.assert_not_called() + assert agent._last_ctx_overflow_warn is None From b656b0d3ba7417a7d4d074236eba1e57fa71061e Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 20 Aug 2026 01:22:00 -0500 Subject: [PATCH 38/42] fix(desktop): the composer surface can no longer be widened by its own content MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The surface is a grid that never declared a column, and an implicit auto column sizes to its items' min-content. The coding status row (branch, PR chip, worktree path, counts — none of it wrapping) out-measured narrow panes and silently set the track wider than the surface; every w-full child laid out against that phantom width and overflow-hidden clipped the right edge, send button first. grid-cols-[minmax(0,1fr)] pins the track to the surface. --- apps/desktop/src/app/chat/composer/index.tsx | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index fc4b05937f..98830fc672 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -321,7 +321,7 @@ export function ChatBar({ return onCancel() }, [activeQueueSessionKeyRef, onCancel]) - const { compactPill, stacked } = useComposerMetrics({ + const { compactPill, foldVoice, minimal, stacked } = useComposerMetrics({ composerDockRef, composerRef, composerSurfaceRef, @@ -984,7 +984,9 @@ export function ChatBar({ status: conversation.status }} disabled={disabled} + foldVoice={foldVoice} hasComposerPayload={hasComposerPayload} + minimal={minimal} onDictate={dictate} onQueue={queueDraft} onToggleAutoSpeak={handleToggleAutoSpeak} @@ -1242,7 +1244,13 @@ export function ChatBar({ {hudMode && busy && }
{input}
-
+
{controls}
From dc90b1b31df75e021b171145e9771bacce391b76 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Thu, 20 Aug 2026 01:22:00 -0500 Subject: [PATCH 39/42] feat(desktop): the composer collapse ladder continues below stacking MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Stacking was the ladder's last rung, but a tile can be dragged far narrower than the stacked controls row costs. Two more width stages, from the same measured-width engine: under 260 the three voice toggles fold into the one menu HUD mode already uses, and under 180 the model pill and the menu drop too — input and Send, nothing else, down to the 80px pane floor. The model pill also shrinks and truncates between stages instead of holding its width, and the metrics hook returns one ComposerFit so a resize re-renders only when a stage actually flips. --- .../src/app/chat/composer/composer-utils.ts | 18 +++++ .../src/app/chat/composer/controls.test.tsx | 31 ++++++++ .../src/app/chat/composer/controls.tsx | 53 ++++++++----- .../composer/hooks/use-composer-metrics.ts | 77 ++++++++++++------- .../src/app/chat/composer/model-pill.tsx | 5 +- 5 files changed, 136 insertions(+), 48 deletions(-) diff --git a/apps/desktop/src/app/chat/composer/composer-utils.ts b/apps/desktop/src/app/chat/composer/composer-utils.ts index 617e650511..3272835892 100644 --- a/apps/desktop/src/app/chat/composer/composer-utils.ts +++ b/apps/desktop/src/app/chat/composer/composer-utils.ts @@ -22,6 +22,24 @@ export const COMPOSER_STACK_BREAKPOINT_PX = 320 // chevron frees is spent keeping the row single for another stretch. export const COMPOSER_COMPACT_PILL_PX = 560 +// The ladder keeps going below the stack breakpoint — a pane can be far +// narrower than even the stacked controls row. Both rungs are budgeted +// against that row's real cost: menu ~24 + surface padding 16 + the cluster +// (~190; ~218 mid-turn with the queue button). +// +// At 260 the three voice toggles fold into the one menu HUD mode already +// uses, clearing the mid-turn worst case with margin. Each stage sits clear +// of the floor below it rather than arriving the instant the previous one +// gives out — the mistake COMPOSER_COMPACT_PILL_PX documents. +export const COMPOSER_FOLD_VOICE_PX = 260 + +// Type and send, nothing else. A pane can be dragged to MIN_PANE_PX (80), and +// even with voice folded the row still costs ~150, so the last rung drops the +// pill AND the voice menu. Both stay reachable — the model by hotkey and the +// full picker, dictation from any wider pane — and Send fits with room to +// spare at any width the layout tree allows (~74 all-in). +export const COMPOSER_MINIMAL_PX = 180 + // A single editor line is ~28px (--composer-input-min-height 1.625rem + 0.5rem // vertical padding). Anything taller means the text wrapped to a second line, // which is when the composer should expand to the stacked layout. diff --git a/apps/desktop/src/app/chat/composer/controls.test.tsx b/apps/desktop/src/app/chat/composer/controls.test.tsx index 7e90654477..bbee021649 100644 --- a/apps/desktop/src/app/chat/composer/controls.test.tsx +++ b/apps/desktop/src/app/chat/composer/controls.test.tsx @@ -98,6 +98,37 @@ describe('HUD mode', () => { }) }) +// A tile can be narrower than the controls cost, and the row is inside an +// overflow-hidden surface — so anything that doesn't fold gets clipped off the +// right edge, send button first. The ladder keeps going past `stacked`: voice +// folds into the same menu the HUD uses, then the model pill drops. Send is +// the last thing standing. +describe('narrow tiles', () => { + it('folds the voice controls into one menu without entering HUD mode', () => { + renderControls({ foldVoice: true }) + + expect(screen.getByLabelText('Voice')).toBeTruthy() + expect(screen.queryByLabelText('Voice dictation')).toBeNull() + expect(screen.queryByLabelText('Read replies aloud')).toBeNull() + + // Folding is a width decision, not the HUD: no exit affordance appears. + expect(screen.queryByLabelText('Exit HUD mode')).toBeNull() + }) + + it('keeps Send at the tightest width, with everything else dropped', () => { + renderControls({ foldVoice: true, minimal: true }) + + expect(screen.getByLabelText('Send')).toBeTruthy() + expect(screen.queryByLabelText('Voice')).toBeNull() + }) + + it('keeps Stop reachable mid-turn at the tightest width', () => { + renderControls({ busy: true, busyAction: 'stop', foldVoice: true, hasComposerPayload: false, minimal: true }) + + expect(screen.getByLabelText('Stop')).toBeTruthy() + }) +}) + describe('ComposerControls shortcut tooltips', () => { it('shows Enter for Send', async () => { renderControls() diff --git a/apps/desktop/src/app/chat/composer/controls.tsx b/apps/desktop/src/app/chat/composer/controls.tsx index 1682f8d9c2..c48e3eb21b 100644 --- a/apps/desktop/src/app/chat/composer/controls.tsx +++ b/apps/desktop/src/app/chat/composer/controls.tsx @@ -39,7 +39,9 @@ export function ComposerControls({ compactModelPill = false, conversation, disabled, + foldVoice = false, hasComposerPayload, + minimal = false, state, voiceStatus, onDictate, @@ -53,7 +55,9 @@ export function ComposerControls({ compactModelPill?: boolean conversation: ConversationProps disabled: boolean + foldVoice?: boolean hasComposerPayload: boolean + minimal?: boolean state: ChatBarState voiceStatus: VoiceStatus onDictate: () => void @@ -73,29 +77,38 @@ export function ComposerControls({ // only when the composer is empty and a turn is running. const showStop = busy && !hasComposerPayload const showQueueButton = busyAction !== 'stop' && hasComposerPayload + // The HUD is a Spotlight bar a few hundred pixels wide, so the four separate + // voice toggles fold into one menu there and leave the row to the input. A + // narrow tile hits the same wall from the other direction and folds for the + // same reason — same controls, same state, different budget. Below that + // even the menu goes: at `minimal` the row is the send button and nothing + // else, which is the one thing that must survive every width. + const foldedVoice = hudMode || foldVoice + + const voiceControls = foldedVoice ? ( + + ) : ( + <> + + + + + ) return ( -
- - {/* The HUD is a Spotlight bar a few hundred pixels wide, so the four - separate voice toggles fold into one menu there and leave the row to - the input. The docked composer has the width and keeps them inline — - same controls, same state, different budget. */} - {hudMode ? ( - - ) : ( +
+ {minimal ? null : ( <> - - - + + {voiceControls} )} {showQueueButton ? ( diff --git a/apps/desktop/src/app/chat/composer/hooks/use-composer-metrics.ts b/apps/desktop/src/app/chat/composer/hooks/use-composer-metrics.ts index 8c09b2f45f..9af4e1be7c 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-composer-metrics.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-composer-metrics.ts @@ -10,7 +10,13 @@ import { } from '@/app/chat/surface-vars' import { useResizeObserver } from '@/hooks/use-resize-observer' -import { COMPOSER_COMPACT_PILL_PX, COMPOSER_SINGLE_LINE_MAX_PX, COMPOSER_STACK_BREAKPOINT_PX } from '../composer-utils' +import { + COMPOSER_COMPACT_PILL_PX, + COMPOSER_FOLD_VOICE_PX, + COMPOSER_MINIMAL_PX, + COMPOSER_SINGLE_LINE_MAX_PX, + COMPOSER_STACK_BREAKPOINT_PX +} from '../composer-utils' interface UseComposerMetricsArgs { composerDockRef: RefObject @@ -20,13 +26,36 @@ interface UseComposerMetricsArgs { poppedOut: boolean } +/** Every width-driven collapse stage, resolved from the composer's own width. */ +export interface ComposerFit { + compactPill: boolean + foldVoice: boolean + minimal: boolean + tight: boolean +} + +const ROOMY: ComposerFit = { compactPill: false, foldVoice: false, minimal: false, tight: false } + +const fitForWidth = (width: number): ComposerFit => ({ + compactPill: width < COMPOSER_COMPACT_PILL_PX, + foldVoice: width < COMPOSER_FOLD_VOICE_PX, + minimal: width < COMPOSER_MINIMAL_PX, + tight: width < COMPOSER_STACK_BREAKPOINT_PX +}) + +const sameFit = (a: ComposerFit, b: ComposerFit) => + a.compactPill === b.compactPill && a.foldVoice === b.foldVoice && a.minimal === b.minimal && a.tight === b.tight + +interface UseComposerMetricsResult extends ComposerFit { + stacked: boolean +} + /** * Owns the composer's *sizing* engine: the stacked-vs-inline layout decision * and the measured-height CSS vars the thread reads for bottom clearance. All * work is edge-gated — the ResizeObserver only fires on real size changes, the * height vars are 8px-bucketed so per-keystroke growth never invalidates the - * tree's computed style, and `tight` only flips when it crosses the breakpoint. - * Returns `stacked` (the only value the render needs). + * tree's computed style, and the fit only re-renders when it crosses a stage. */ export function useComposerMetrics({ composerDockRef, @@ -34,14 +63,9 @@ export function useComposerMetrics({ composerSurfaceRef, editorRef, poppedOut -}: UseComposerMetricsArgs): { - compactPill: boolean - stacked: boolean -} { +}: UseComposerMetricsArgs): UseComposerMetricsResult { const [expanded, setExpanded] = useState(false) - const [tight, setTight] = useState(false) - // Wider than `tight`: the pill goes icon-only before the row has to stack. - const [compactPill, setCompactPill] = useState(false) + const [fit, setFit] = useState(ROOMY) // Edge signals, not the live text: these only re-render when emptiness / the // presence of a non-trailing newline actually flips, so typing within a line @@ -84,8 +108,7 @@ export function useComposerMetrics({ // until a wrap or row change actually happens. const lastBucketedHeightRef = useRef(0) const lastBucketedSurfaceHeightRef = useRef(0) - const lastTightRef = useRef(null) - const lastCompactPillRef = useRef(null) + const lastFitRef = useRef(ROOMY) // Mirrored into a ref so `syncComposerMetrics` stays referentially stable — // it's the shared ResizeObserver's handler, and a new identity every render // would re-register the observation. @@ -121,18 +144,11 @@ export function useComposerMetrics({ const surfaceHeight = composerSurfaceRef.current?.getBoundingClientRect().height if (width > 0) { - const nextTight = width < COMPOSER_STACK_BREAKPOINT_PX + const nextFit = fitForWidth(width) - if (nextTight !== lastTightRef.current) { - lastTightRef.current = nextTight - setTight(nextTight) - } - - const nextCompactPill = width < COMPOSER_COMPACT_PILL_PX - - if (nextCompactPill !== lastCompactPillRef.current) { - lastCompactPillRef.current = nextCompactPill - setCompactPill(nextCompactPill) + if (!sameFit(nextFit, lastFitRef.current)) { + lastFitRef.current = nextFit + setFit(nextFit) } } @@ -189,7 +205,7 @@ export function useComposerMetrics({ } }, [composerRef]) - // Both decisions come from the composer's OWN measured width, never the + // Every decision comes from the composer's OWN measured width, never the // viewport's. There used to be a `(max-width: 30rem)` media query in here as // well, and it quietly outranked everything: any window under 480px stacked // the row AND compacted the pill in the same instant, regardless of how much @@ -199,7 +215,14 @@ export function useComposerMetrics({ // stack) by 160px. The ResizeObserver knows the real width; the viewport is // not a proxy for it. // - // The pill still compacts whenever the row stacks, so the controls row can't - // over-run once it has the width to itself. - return { compactPill: compactPill || tight, stacked: expanded || tight } + // The ladder is monotonic: each stage implies the ones above it, so the pill + // is always compact by the time the row stacks, and the voice controls are + // always folded before minimal drops them. + return { + compactPill: fit.compactPill || fit.tight, + foldVoice: fit.foldVoice || fit.minimal, + minimal: fit.minimal, + stacked: expanded || fit.tight, + tight: fit.tight + } } diff --git a/apps/desktop/src/app/chat/composer/model-pill.tsx b/apps/desktop/src/app/chat/composer/model-pill.tsx index e41e318928..6bf8c9f0ae 100644 --- a/apps/desktop/src/app/chat/composer/model-pill.tsx +++ b/apps/desktop/src/app/chat/composer/model-pill.tsx @@ -18,8 +18,11 @@ import { onComposerModelMenuRequest } from './focus' import { useComposerScope } from './scope' import type { ChatBarState } from './types' +// `shrink` (not `shrink-0`) with a truncating label: the pill is the one +// control in the row that can give width back continuously, so it absorbs the +// squeeze between collapse stages instead of pushing Send past the edge. const PILL = cn( - 'h-(--composer-control-size) max-w-40 shrink-0 gap-1 rounded-md px-2 text-xs font-normal', + 'h-(--composer-control-size) min-w-0 max-w-40 shrink gap-1 rounded-md px-2 text-xs font-normal', 'text-(--ui-text-tertiary) hover:bg-(--chrome-action-hover) hover:text-foreground' ) From ad7a14a53930f61efe0a7dc6c110ff0e2ddbe897 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 23:23:05 -0700 Subject: [PATCH 40/42] fix(desktop): renamed Bot Mode agents stay @-taggable by their new name MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Renaming a bot (Bot Mode title or 'hermes profile rename' display_name) changed the roster row but not what the user could @-tag it with — mentions still only resolved the original profile handle, and the composer autocomplete never offered the new name. - mentionNameForms()/botFriendlyNames()/botMentionTag(): one resolver for the taggable forms a friendly name yields (slugged + collapsed), with reserved tokens (hermes/default/everyone/all/user) excluded so a rename can never hijack them. - resolveRosterMentions() and parseGroupChatMentions() accept the friendly forms alongside the profile name/handle (both keep working). - Composer @ autocomplete (global provider + group-room popover) inserts the renamed tag and prefix-matches on tag, handle, and display name. - Mention middleware's cold-cache fallback now runs the same resolver instead of a bare-names-only parse, so renamed tags resolve there too. - durableGroupChatMembers persists title/display_name so renamed-tag mentions survive connection switches in cross-machine rooms. - Docs: bot-mode.md documents renamed tags. --- .../desktop/src/plugins/hermes-bots/plugin.js | 148 ++++++++++---- .../tests/mention-renamed-bots.test.mjs | 188 ++++++++++++++++++ website/docs/user-guide/bot-mode.md | 1 + 3 files changed, 302 insertions(+), 35 deletions(-) create mode 100644 apps/desktop/src/plugins/hermes-bots/tests/mention-renamed-bots.test.mjs diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index 88401d2a94..b8d3248ce5 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -2969,6 +2969,53 @@ function botHandle(name, bot) { return (name || '').trim().toLowerCase() === 'default' ? 'hermes' : name } +/** Taggable @-forms derived from a bot's friendly names — the core profile + * display name (`hermes profile rename`) and the Bot Mode title. Free text + * reduces to the mention charset two ways: slugified ("Research Buddy" → + * research-buddy, the form autocomplete inserts) and collapsed + * (researchbuddy). Reserved tokens are dropped so a bot renamed "Hermes" + * can never hijack the primary profile's @hermes alias. */ +function mentionNameForms(value) { + const name = String(value || '').trim().toLowerCase() + + if (!name) { + return [] + } + + const slug = name.replace(/[^a-z0-9_-]+/g, '-').replace(/^-+|-+$/g, '') + const collapsed = name.replace(/[^a-z0-9_-]+/g, '') + + return [...new Set([slug, collapsed])].filter( + form => /^[a-z0-9][a-z0-9_-]*$/.test(form) && !['all', 'everyone', 'user', 'default', 'hermes'].includes(form) + ) +} + +/** Every friendly (renameable) name a roster row carries: the Bot Mode title + * (server-synced via ui_meta, locally stored, or persisted on a durable + * group descriptor) and the core profile display_name — in displayName's + * precedence order. Remote rows never borrow local meta (two `default`s + * must not share a title). */ +function botFriendlyNames(bot) { + const localTitle = !bot?.remoteSource && typeof $botMeta !== 'undefined' ? $botMeta.get()?.[bot?.name]?.title : null + + return [bot?.ui_meta?.['hermes-bots']?.title, localTitle, bot?.title, bot?.display_name] +} + +/** The tag autocomplete inserts for a bot: the renamed (friendly) slug when + * the user gave the bot a real name, otherwise the profile @handle. The + * resolvers accept both, so older muscle memory keeps working. */ +function botMentionTag(bot) { + for (const friendly of botFriendlyNames(bot)) { + const forms = mentionNameForms(friendly) + + if (forms.length) { + return forms[0] + } + } + + return botHandle(bot?.name, bot) +} + function isActiveRosterBot(bot, active) { const activeName = String(active?.name || 'default').trim() || 'default' const activeId = String(active?.connectionId || '').trim() @@ -3007,6 +3054,15 @@ function resolveRosterMentions(text, roster, active = {}) { forms.add(String(bot.handle).toLowerCase()) } + // Renamed bots are taggable by their friendly names too — the core + // profile display_name and the Bot Mode title (issue: renaming a bot + // didn't change what you @-tag it with). + for (const friendly of botFriendlyNames(bot)) { + for (const form of mentionNameForms(friendly)) { + forms.add(form) + } + } + for (const form of forms) { if (!form) { continue @@ -3697,15 +3753,24 @@ function groupChatMemberBots(group, roster, metaByName) { * source's row may become remote after a connection switch, so retaining it * here is what keeps the same room intact across machines. */ function durableGroupChatMembers(bots) { - return (bots || []).map(bot => ({ - name: bot.name, - handle: bot.handle || bot.name, - connectionId: bot.connectionId, - connectionKind: bot.connectionKind, - connectionLabel: bot.connectionLabel, - remoteSource: true, - sourceScoped: true - })) + return (bots || []).map(bot => { + // Keep the friendly identity on the stored descriptor: after a + // connection switch the live roster row may be gone, and renamed-tag + // mentions must still resolve against the persisted member. + const title = String(botRosterMeta(bot, $botMeta.get())?.title || bot.ui_meta?.['hermes-bots']?.title || bot.title || '').trim() + + return { + name: bot.name, + handle: bot.handle || bot.name, + ...(title ? { title } : {}), + ...(bot.display_name ? { display_name: bot.display_name } : {}), + connectionId: bot.connectionId, + connectionKind: bot.connectionKind, + connectionLabel: bot.connectionLabel, + remoteSource: true, + sourceScoped: true + } + }) } /** Existing group names, alphabetical — feeds the Manage-groups dialog. */ @@ -3774,6 +3839,15 @@ function parseGroupChatMentions(text, members) { : []) ]) + // Renamed members answer to their friendly names too (profile + // display_name and Bot Mode title), in slugged and collapsed forms — + // the same tags the roster autocomplete inserts. + for (const friendly of botFriendlyNames(member)) { + for (const form of mentionNameForms(friendly)) { + forms.add(form) + } + } + for (const form of forms) { if (form) { handles.set(form, groupMemberKey(member)) @@ -8658,14 +8732,26 @@ function GroupMentionInput({ members, onChange, value, ...inputProps }) { for (const member of members) { const handle = String(member.handle || botHandle(member.name, member) || '').trim() + const display = displayName(member, botRosterMeta(member, allMeta)) + // Renamed members complete on their friendly tag; parser resolves both. + const tag = String(botMentionTag(member) || handle).trim() - if (!handle || (token.query && !handle.toLowerCase().startsWith(token.query))) { + if (!tag) { + continue + } + + if ( + token.query && + !tag.toLowerCase().startsWith(token.query) && + !(handle && handle.toLowerCase().startsWith(token.query)) && + !display.toLowerCase().startsWith(token.query) + ) { continue } options.push({ - handle, - meta: displayName(member, botRosterMeta(member, allMeta)) + handle: tag, + meta: display }) } } @@ -10109,17 +10195,25 @@ export default { } const handle = botHandle(profile.name, profile) + const display = displayName(profile, $botMeta.get()[profile.name]) + // Renamed bots complete on their friendly name — the tag is the + // renamed slug when one exists, the profile handle otherwise. + const tag = botMentionTag(profile) - if (q && !handle.toLowerCase().startsWith(q)) { + if ( + q && + !tag.toLowerCase().startsWith(q) && + !handle.toLowerCase().startsWith(q) && + !display.toLowerCase().startsWith(q) + ) { continue } - const display = displayName(profile, $botMeta.get()[profile.name]) const source = profile.connectionLabel ? ` · ${profile.connectionLabel}` : '' items.push({ - insert: `@${handle}`, - display: `@${handle}`, + insert: `@${tag}`, + display: `@${tag}`, meta: `Bot · ${display}${source}` }) } @@ -10397,30 +10491,14 @@ export default { let mentionedBots = roster ? resolveRosterMentions(text, roster, live) : [] if (!roster) { - let names = [] try { const res = await host.request('profiles.list', { include_sessions: false }) - names = (res?.profiles ?? []).map(p => p.name) + // Same resolver as the cached path — renamed bots (display_name + // / ui_meta title) stay taggable when the roster cache is cold. + mentionedBots = resolveRosterMentions(text, res?.profiles ?? [], live).map(bot => ({ ...bot, remoteSource: false })) } catch { return draft } - - const prose = text.replace(/```[\s\S]*?```/g, ' ').replace(/`[^`\n]*`/g, ' ') - const mentioned = [] - - for (const match of prose.matchAll(/(^|\s)@([a-z0-9][a-z0-9_-]*)/gi)) { - let name = match[2].toLowerCase() - - if (name === 'hermes' && !names.includes('hermes') && names.includes('default')) { - name = 'default' - } - - if (names.includes(name) && name !== live.name && !mentioned.includes(name)) { - mentioned.push(name) - } - } - - mentionedBots = mentioned.map(name => ({ name })) } if (!mentionedBots.length) { diff --git a/apps/desktop/src/plugins/hermes-bots/tests/mention-renamed-bots.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/mention-renamed-bots.test.mjs new file mode 100644 index 0000000000..afa10e4746 --- /dev/null +++ b/apps/desktop/src/plugins/hermes-bots/tests/mention-renamed-bots.test.mjs @@ -0,0 +1,188 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import vm from 'node:vm' + +// Renamed bots stay taggable (Discord report, Aug 2026): renaming a bot — +// Bot Mode title or `hermes profile rename` display_name — must update what +// the user can @-tag it with, in the mention middleware, group rooms, and +// the composer autocomplete. Old profile handles keep resolving. + +const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') + +function runtime({ meta } = {}) { + const context = { + console, + setTimeout, + clearTimeout, + Date, + URL, + atom: initial => { + let value = initial + return { get: () => value, set: next => (value = next), listen: () => () => undefined } + }, + host: { + request: async () => ({}), + requestProfile: async () => ({}), + state: { + profile: { get: () => 'default', listen: () => undefined }, + connectionId: { get: () => 'local', listen: () => undefined }, + gateway: { listen: () => undefined } + } + }, + document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } } + } + const code = source + .replace(/^import\s+\*\s+as\s+sdk\s+from '@hermes\/plugin-sdk'\r?\n/m, '') + .replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '') + .replace(/^const \{ McpTab, ToolsetConfigPanel \} = sdk\r?\n/m, '') + .replace(/^import .* from 'react'\r?\n/m, '') + .replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '') + .replace('export default {', 'globalThis.plugin = {') + .concat( + '\nglobalThis.__x = { mentionNameForms, botFriendlyNames, botMentionTag, resolveRosterMentions, parseGroupChatMentions, groupMemberKey, durableGroupChatMembers, $botMeta };\n' + ) + vm.runInNewContext(code, context, { filename: 'plugin.js' }) + if (meta) { + context.__x.$botMeta.set(meta) + } + return context.__x +} + +test('mentionNameForms: slugged + collapsed, reserved tokens dropped', () => { + const { mentionNameForms } = runtime() + + assert.deepEqual(JSON.parse(JSON.stringify(mentionNameForms('Research Buddy'))), ['research-buddy', 'researchbuddy']) + assert.deepEqual(JSON.parse(JSON.stringify(mentionNameForms('Ops'))), ['ops']) + // A bot renamed "Hermes" or "everyone" cannot hijack reserved tags. + assert.equal(mentionNameForms('Hermes').length, 0) + assert.equal(mentionNameForms('@everyone').length, 0) + assert.equal(mentionNameForms('').length, 0) +}) + +test('botMentionTag: renamed slug wins, profile handle is the fallback', () => { + const { botMentionTag } = runtime({ meta: { writer: { title: 'Research Buddy' } } }) + + assert.equal(botMentionTag({ name: 'writer' }), 'research-buddy') + assert.equal(botMentionTag({ name: 'ops' }), 'ops') + assert.equal(botMentionTag({ name: 'default' }), 'hermes') + // display_name (hermes profile rename) drives the tag too. + assert.equal(botMentionTag({ name: 'scout', display_name: 'Deal Finder' }), 'deal-finder') +}) + +test('resolveRosterMentions: renamed bots resolve by friendly tag AND old handle', () => { + const { resolveRosterMentions } = runtime({ meta: { writer: { title: 'Research Buddy' } } }) + const roster = [{ name: 'default' }, { name: 'writer' }, { name: 'scout', display_name: 'Deal Finder' }] + const live = { name: 'default', connectionId: 'local' } + + const byTitle = resolveRosterMentions('hey @research-buddy check this', roster, live) + assert.deepEqual(JSON.parse(JSON.stringify(byTitle.map(b => b.name))), ['writer']) + + const byDisplayName = resolveRosterMentions('ping @dealfinder please', roster, live) + assert.deepEqual(JSON.parse(JSON.stringify(byDisplayName.map(b => b.name))), ['scout']) + + // The profile name keeps working after a rename. + const byName = resolveRosterMentions('hey @writer', roster, live) + assert.deepEqual(JSON.parse(JSON.stringify(byName.map(b => b.name))), ['writer']) +}) + +test('parseGroupChatMentions: display_name and ui_meta title forms address a member', () => { + const { parseGroupChatMentions, groupMemberKey } = runtime() + const members = [ + { name: 'research', title: '' }, + { name: 'builder', title: '', display_name: 'Site Smith' }, + { name: 'ops', title: '', ui_meta: { 'hermes-bots': { title: 'Night Watch' } } } + ] + + const bySlug = parseGroupChatMentions('@site-smith and @night-watch please', members) + assert.equal(bySlug.mentioned.has(groupMemberKey(members[1])), true) + assert.equal(bySlug.mentioned.has(groupMemberKey(members[2])), true) + assert.equal(bySlug.mentioned.has(groupMemberKey(members[0])), false) + + const collapsed = parseGroupChatMentions('@sitesmith take a look', members) + assert.equal(collapsed.mentioned.has(groupMemberKey(members[1])), true) +}) + +test('durableGroupChatMembers persists the friendly identity for renamed-tag mentions', () => { + const { durableGroupChatMembers, $botMeta } = runtime() + $botMeta.set({ writer: { title: 'Research Buddy' } }) + + const [stored] = durableGroupChatMembers([{ name: 'writer', connectionId: 'local' }]) + assert.equal(stored.title, 'Research Buddy') + + const [remote] = durableGroupChatMembers([ + { name: 'scout', display_name: 'Deal Finder', connectionId: 'mac-mini', remoteSource: true } + ]) + assert.equal(remote.display_name, 'Deal Finder') +}) + +test('composer autocomplete offers the renamed tag and matches on the display name', () => { + const registrations = [] + const atom = value => { + let current = value + return { get: () => current, set: next => (current = next), listen: () => () => undefined } + } + const jsx = (type, props = {}) => ({ type, props }) + const roster = { profiles: [{ name: 'default' }, { name: 'writer' }, { name: 'ops' }] } + const context = { + atom, + jsx, + jsxs: jsx, + useQuery: () => ({}), + useValue: v => (v?.get ? v.get() : v), + useState: v => [v, () => undefined], + setTimeout, + clearTimeout, + performance, + window: { + setTimeout, + clearTimeout, + performance, + requestAnimationFrame: () => 0, + cancelAnimationFrame: () => undefined, + addEventListener: () => undefined, + removeEventListener: () => undefined + }, + document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } }, + queryClient: { getQueryData: () => roster, invalidateQueries: () => undefined, setQueryData: () => undefined }, + host: { + state: { profile: { get: () => 'default', listen: () => () => undefined }, gateway: { get: () => null, listen: () => () => undefined } }, + request: async () => ({}), + onEvent: () => () => undefined + }, + COMPOSER_AREAS: { middleware: 'composer.middleware', atCompletions: 'composer.atCompletions' }, + sdk: new Proxy({}, { get: () => undefined }) + } + const code = source + .replace(/^import \* as sdk from '@hermes\/plugin-sdk'\r?\n/m, '') + .replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '') + .replace(/^import .* from 'react'\r?\n/m, '') + .replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '') + .replace('export default {', 'globalThis.plugin = {') + .concat('\nglobalThis.__botMeta = $botMeta;') + vm.runInNewContext(code, context) + context.__botMeta.set({ writer: { title: 'Research Buddy' } }) + + try { + context.plugin.register({ + register: c => registrations.push(c), + storage: { get: async () => undefined, set: async () => undefined } + }) + } catch { + /* registration walks UI surfaces the stub doesn't fully model */ + } + + const provide = registrations.find(c => c.area === 'composer.atCompletions').data.provide + + // The renamed bot completes under its friendly prefix and inserts the tag. + const renamed = provide('rese') + assert.deepEqual(JSON.parse(JSON.stringify(renamed.map(i => i.insert))), ['@research-buddy']) + + // Old profile-handle muscle memory still finds it (insert is the new tag). + const byHandle = provide('wri') + assert.deepEqual(JSON.parse(JSON.stringify(byHandle.map(i => i.insert))), ['@research-buddy']) + + // Un-renamed bots keep their plain handle. + const plain = provide('op') + assert.deepEqual(JSON.parse(JSON.stringify(plain.map(i => i.insert))), ['@ops']) +}) diff --git a/website/docs/user-guide/bot-mode.md b/website/docs/user-guide/bot-mode.md index 9b7395c65d..032f64003a 100644 --- a/website/docs/user-guide/bot-mode.md +++ b/website/docs/user-guide/bot-mode.md @@ -94,6 +94,7 @@ Groups are standalone rows in the same activity-ordered roster as Bot DMs. A Bot Bots message each other with attribution, and you can hand work off from any chat: - **@mentions** — type `@researcher have a look at this` in any chat and the active Bot hands the message off, waits for the reply, and reports back. Mention names are validated against the live roster, so an email address or an unknown `@` passes through untouched. +- **Renamed Bots keep their tags in sync** — give a Bot a friendly name (the pencil in its chat header, or `hermes profile rename`) and it becomes taggable by that name: a Bot titled *Research Buddy* answers to `@research-buddy` (and `@researchbuddy`), in regular chats and in group rooms alike. The composer's `@` autocomplete offers the renamed tag and also matches when you type the old profile name, which keeps resolving too. - **@mentions across machines** — mentioning a Bot that lives on another registered connection (use its `@name-device` handle when names collide) delivers over the Connections registry in the background: the active Bot stays on this device, the desktop routes the message to the recipient's machine, and the reply is relayed back attributed to that agent. Your window's gateway never switches. - **Direct messages** — a Bot reaches a teammate's Bot Chat through the standard CLI: it writes the message to a temp file (opening with the `Message from 🤖 (@):` prefix), then runs `hermes -p chat --in ~ -c "Bot Chat" --create-if-missing -Q --query-file `. The file transport means nothing is shell-interpreted — quotes, `$(...)`, and backticks in the message arrive verbatim. The receiving Bot sees the message the next time it runs and knows how to reply, because the messaging protocol is part of its Bot Chat system prompt. From 761990b7800bf130a5a22241ec8e76168cae28d2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 23:29:28 -0700 Subject: [PATCH 41/42] feat: identical re-calls enter context as reference stubs, not duplicate payloads --- agent/tool_executor.py | 23 +++ agent/tool_guardrails.py | 177 +++++++++++++++++++---- run_agent.py | 25 +++- tests/agent/test_stall_guards.py | 171 +++++++++++++++++++++- tools/tool_result_storage.py | 16 ++ website/docs/user-guide/configuration.md | 2 + 6 files changed, 382 insertions(+), 32 deletions(-) diff --git a/agent/tool_executor.py b/agent/tool_executor.py index 2eb69c6124..4a828477c3 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -48,12 +48,31 @@ from tools.thread_context import propagate_context_to_thread from tools.tool_result_storage import ( maybe_persist_tool_result, enforce_turn_budget, + extract_persisted_path, ) from tools.budget_config import BudgetConfig, DEFAULT_BUDGET, budget_for_context_window logger = logging.getLogger(__name__) +def _record_persisted_path_for_stub(agent, tool_call_id: str, function_result) -> None: + """Tell the stall guards where a persisted result's full content lives. + + When a large result is spilled to disk ( preview), a + later result-reference stub pointing at that first occurrence must carry + the spillover file path so the reference can't dangle. Best-effort: never + lets bookkeeping break tool execution. + """ + try: + if not isinstance(function_result, str): + return + path = extract_persisted_path(function_result) + if path: + agent._tool_guardrails.record_persisted_result(tool_call_id, path) + except Exception as exc: + logger.debug("persisted-path record for result stub failed: %s", exc) + + def _ensure_file_checkpoint( agent, function_name: str, @@ -1733,6 +1752,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe function_args, function_result, failed=is_error, + tool_call_id=getattr(tc, "id", "") or "", ) if is_error: @@ -1767,6 +1787,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe env=get_active_env(effective_task_id), config=_tool_budget, ) if not _is_multimodal_tool_result(function_result) else function_result + _record_persisted_path_for_stub(agent, tc.id, function_result) subdir_hints = agent._subdirectory_hints.check_tool_call(name, args) if subdir_hints: @@ -2572,6 +2593,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe function_args, function_result, failed=_is_error_result, + tool_call_id=getattr(tool_call, "id", "") or "", ) result_preview = function_result if agent.verbose_logging else ( function_result[:200] if len(function_result) > 200 else function_result @@ -2610,6 +2632,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe env=get_active_env(effective_task_id), config=_tool_budget, ) if not _is_multimodal_tool_result(function_result) else function_result + _record_persisted_path_for_stub(agent, tool_call.id, function_result) # Discover subdirectory context files from tool arguments subdir_hints = agent._subdirectory_hints.check_tool_call(function_name, function_args) diff --git a/agent/tool_guardrails.py b/agent/tool_guardrails.py index b319046335..e4afbbba82 100644 --- a/agent/tool_guardrails.py +++ b/agent/tool_guardrails.py @@ -85,6 +85,19 @@ _STALL_GUARD_REPEATABLE_SUFFIXES = ( # catching the observed re-issue loops (3x/4x identical calls in eval traces). STALL_GUARD_IDENTICAL_CALL_THRESHOLD = 3 +# Result-reference stubbing (agent.stall_guards): from the 2nd consecutive +# identical call whose FRESH result is byte-identical to the previous one, +# the duplicate payload is replaced in context by a short reference stub. +# Results under this size aren't worth stubbing (the stub itself plus the +# lost locality outweigh the savings), and error results are never stubbed +# (the model must see every fresh error verbatim). +IDENTICAL_RESULT_STUB_MIN_CHARS = 512 + +# How much of the canonical args JSON the stub carries so the model still +# knows WHAT the referenced call was even if context compression later +# evicts the referenced result (cheap dangling-reference mitigation). +_RESULT_STUB_ARGS_PREVIEW_CHARS = 120 + def is_stall_guard_repeatable(tool_name: str) -> bool: """Whether a tool is exempt from the identical-call loop notice.""" @@ -206,6 +219,21 @@ class LoopCapConfig: ) +@dataclass(frozen=True) +class IdenticalCallObservation: + """Outcome of observing one completed tool call for the stall guards. + + ``notice`` is the identical-call loop-breaker notice (appended after the + result). ``stub`` is the result-reference replacement for a byte-identical + duplicate result (replaces the result content). Both may be set on the + same call (3rd+ identical call): the stub replaces the payload and the + notice is appended after it. + """ + + notice: str | None = None + stub: str | None = None + + @dataclass(frozen=True) class ToolCallSignature: """Stable, non-reversible identity for a tool name plus canonical args.""" @@ -320,9 +348,21 @@ class ToolCallGuardrailController: # results were also identical. Any different call — or a different # result — resets the streak, so legitimate re-reads after edits and # varied polling are never flagged. Per-turn, like everything else here. + # NOTE: open PR #85352 (patrykkopycinski) tracks no-progress loops + # ACROSS turns via a detection window — a different mechanism from + # this per-turn consecutive streak. Coordinate future work there. self._identical_streak_sig: ToolCallSignature | None = None self._identical_streak_result_hash: str = "" self._identical_streak_count: int = 0 + # tool_call_id of the FIRST call in the current streak, so a + # result-reference stub can point at the message that carries the + # full payload. + self._identical_streak_first_call_id: str = "" + # tool_call_id -> spillover file path for results that were persisted + # out of context (persisted-output preview). Lets a reference stub + # carry the file path so the reference can't dangle when the first + # occurrence entered context as a preview. + self._persisted_result_paths: dict[str, str] = {} # Per-turn runaway-loop cap counters. Reset every turn (this method # runs at the start of each run_conversation), so the caps bound a # single agent loop rather than accumulating across the session. @@ -493,44 +533,129 @@ class ToolCallGuardrailController: ) -> str | None: """Track consecutive identical calls; return a loop-breaker notice or None. - Fires the compact notice when the SAME tool is called with identical - canonical arguments AND returns an identical result for the - ``STALL_GUARD_IDENTICAL_CALL_THRESHOLD``-th (and every subsequent) - consecutive time within the turn. Purely observational — never blocks - the call. Allowlisted pollers (``is_stall_guard_repeatable``) are - exempt, and any intervening different call or changed result resets - the streak. Callers append the returned notice to the tool RESULT at - construction time, which is cache-safe: tool results are append-only - and never mutate already-sent context. + Back-compat wrapper around :meth:`observe_call` for callers that only + care about the loop-breaker notice. """ - if is_stall_guard_repeatable(tool_name): - # Don't let a poller streak carry over into the next tool either. - self._identical_streak_sig = None - self._identical_streak_count = 0 - return None + return self.observe_call(tool_name, args, result).notice + def observe_call( + self, + tool_name: str, + args: Mapping[str, Any] | None, + result: str | None, + *, + tool_call_id: str = "", + failed: bool = False, + ) -> "IdenticalCallObservation": + """Track consecutive identical calls; return notice + dedupe stub info. + + Two independent outputs from the same consecutive-streak tracker: + + - ``notice``: the compact loop-breaker notice, fired when the SAME + tool is called with identical canonical arguments AND returns an + identical result for the ``STALL_GUARD_IDENTICAL_CALL_THRESHOLD``-th + (and every subsequent) consecutive time within the turn. Purely + observational — never blocks the call. Allowlisted pollers + (``is_stall_guard_repeatable``) are exempt from the NOTICE. + - ``stub``: a short reference replacement for the CURRENT result, + produced from the 2nd consecutive identical call whose fresh result + is byte-identical to the previous one. The tool still executed — + only the context representation is deduplicated, so polling + semantics are preserved (a changed result flows through whole and + resets the streak). Pollers are NOT exempt from stubbing: for a + poller, an identical result means nothing changed, which is exactly + when the stub saves the most context and loses nothing. Results + under ``IDENTICAL_RESULT_STUB_MIN_CHARS`` and failed/error results + are never stubbed, and only plain-string results are considered. + + Any intervening different call or changed result resets the streak. + Callers substitute/append at tool RESULT construction time, which is + cache-safe: tool results are append-only and never mutate + already-sent context. + """ + is_plain_str = isinstance(result, str) signature = ToolCallSignature.from_call(tool_name, _coerce_args(args)) - result_hash = _result_hash(result) + result_hash = _result_hash(result) if is_plain_str else "" + if ( - self._identical_streak_sig == signature + is_plain_str + and self._identical_streak_sig == signature and self._identical_streak_result_hash == result_hash ): self._identical_streak_count += 1 else: - self._identical_streak_sig = signature + # New streak (or non-string result, which never forms a streak — + # multimodal content lists pass through untouched). + self._identical_streak_sig = signature if is_plain_str else None self._identical_streak_result_hash = result_hash - self._identical_streak_count = 1 + self._identical_streak_count = 1 if is_plain_str else 0 + self._identical_streak_first_call_id = tool_call_id or "" count = self._identical_streak_count - if count < STALL_GUARD_IDENTICAL_CALL_THRESHOLD: - return None - ordinal = f"{count}{'th' if 11 <= count % 100 <= 13 else {1: 'st', 2: 'nd', 3: 'rd'}.get(count % 10, 'th')}" - return ( - f"[hermes note: this is the {ordinal} consecutive identical call to " - f"{tool_name} with identical arguments returning the same result. " - "Do not repeat it — change arguments, use a different tool, or " - "proceed with what you have.]" + + notice = None + if ( + not is_stall_guard_repeatable(tool_name) + and count >= STALL_GUARD_IDENTICAL_CALL_THRESHOLD + ): + ordinal = f"{count}{'th' if 11 <= count % 100 <= 13 else {1: 'st', 2: 'nd', 3: 'rd'}.get(count % 10, 'th')}" + notice = ( + f"[hermes note: this is the {ordinal} consecutive identical call to " + f"{tool_name} with identical arguments returning the same result. " + "Do not repeat it — change arguments, use a different tool, or " + "proceed with what you have.]" + ) + + stub = None + if ( + is_plain_str + and count >= 2 + and not failed + and len(result) >= IDENTICAL_RESULT_STUB_MIN_CHARS + ): + stub = self._build_result_reference_stub(tool_name, args) + + return IdenticalCallObservation(notice=notice, stub=stub) + + def record_persisted_result(self, tool_call_id: str, file_path: str) -> None: + """Remember the spillover path a persisted result was saved to. + + When the first occurrence of a result entered context as a + persisted-output preview, a later reference stub must carry the + spillover file path so the reference can't dangle. + """ + if tool_call_id and file_path: + self._persisted_result_paths[tool_call_id] = file_path + + def _build_result_reference_stub( + self, tool_name: str, args: Mapping[str, Any] | None + ) -> str: + """Build the reference stub replacing a byte-identical duplicate result. + + Carries the tool name + a canonical-args preview so that even if + context compression later evicts the referenced result, the model + still knows WHAT the call was (cheap dangling-reference mitigation). + """ + try: + args_preview = canonical_tool_args(_coerce_args(args)) + except TypeError: + args_preview = "{}" + if len(args_preview) > _RESULT_STUB_ARGS_PREVIEW_CHARS: + args_preview = args_preview[:_RESULT_STUB_ARGS_PREVIEW_CHARS] + "…" + first_id = self._identical_streak_first_call_id + ref = f" (tool_call_id {first_id})" if first_id else "" + stub = ( + f"[hermes note: this result is byte-identical to the {tool_name} " + f"result earlier this turn{ref}. Refer to that result; it has not " + f"changed. Args: {args_preview}]" ) + spill_path = self._persisted_result_paths.get(first_id) if first_id else None + if spill_path: + stub += ( + f"\n[The referenced result was persisted to: {spill_path} — " + "page through it with read_file if you need the full content.]" + ) + return stub def _check_loop_cap( self, diff --git a/run_agent.py b/run_agent.py index ac06aafaf1..a54bc9cf8f 100644 --- a/run_agent.py +++ b/run_agent.py @@ -8239,6 +8239,7 @@ class AIAgent: function_result: str, *, failed: bool, + tool_call_id: str = "", ) -> str: decision = self._tool_guardrails.after_call( tool_name, @@ -8246,20 +8247,36 @@ class AIAgent: function_result, failed=failed, ) - # Identical-call loop breaker (agent.stall_guards): notice-only, no + # Identical-call stall guards (agent.stall_guards): notice-only, no # blocking. Observed on the RAW result (before the loop-warning suffix # below, whose embedded count changes per call and would defeat - # result-identity matching). Appended here — at result construction, + # result-identity matching). Applied here — at result construction, # before the tool message is built — so it is cache-safe (tool results # are append-only; nothing already sent to the provider is mutated). stall_notice = None + result_stub = None if self._stall_guards_enabled(): try: - stall_notice = self._tool_guardrails.observe_identical_call( - tool_name, function_args, function_result, + observation = self._tool_guardrails.observe_call( + tool_name, + function_args, + function_result if isinstance(function_result, str) else None, + tool_call_id=tool_call_id, + failed=failed, ) + stall_notice = observation.notice + result_stub = observation.stub except Exception as exc: logger.debug("stall-guard identical-call observation failed: %s", exc) + # Result-reference stubbing: a 2nd+ consecutive identical call whose + # FRESH result is byte-identical enters context as a short reference + # stub instead of the duplicate payload. The tool still executed — + # this is not a cache; a changed result flows through whole. Only + # plain-string results are stubbed (multimodal content lists pass + # through untouched), and the current message keeps its role and + # tool_call_id — only the content is replaced. + if result_stub and isinstance(function_result, str): + function_result = result_stub if decision.action in {"warn", "halt"}: function_result = append_toolguard_guidance(function_result, decision) if decision.should_halt: diff --git a/tests/agent/test_stall_guards.py b/tests/agent/test_stall_guards.py index 33267362c0..4ff12fe56e 100644 --- a/tests/agent/test_stall_guards.py +++ b/tests/agent/test_stall_guards.py @@ -17,6 +17,7 @@ These assert behavior contracts, not message snapshots. from agent.agent_runtime_helpers import trailing_continue_intent from agent.tool_guardrails import ( + IDENTICAL_RESULT_STUB_MIN_CHARS, STALL_GUARD_IDENTICAL_CALL_THRESHOLD, STALL_GUARD_REPEATABLE_TOOLS, ToolCallGuardrailController, @@ -142,8 +143,10 @@ def _fake_agent(stall_guards=True): lambda decision: AIAgent._set_tool_guardrail_halt(agent, decision) ) agent._append = ( - lambda name, args, result: AIAgent._append_guardrail_observation( - agent, name, args, result, failed=False + lambda name, args, result, failed=False, tool_call_id="": ( + AIAgent._append_guardrail_observation( + agent, name, args, result, failed=failed, tool_call_id=tool_call_id + ) ) ) return agent @@ -181,6 +184,170 @@ def test_notice_streak_keys_on_raw_result_not_annotated_result(): assert "hermes note" in outs[2] +# ── result-reference stubbing (byte-identical duplicate results) ────────── + + +_BIG = "x" * IDENTICAL_RESULT_STUB_MIN_CHARS # exactly at the stub threshold + + +def test_stub_on_second_identical_call_first_full(): + agent = _fake_agent() + args = {"query": "hermes"} + r1 = agent._append("web_search", args, _BIG, tool_call_id="call_1") + r2 = agent._append("web_search", args, _BIG, tool_call_id="call_2") + assert r1 == _BIG # first occurrence always enters context whole + assert r2 != _BIG + assert "byte-identical" in r2 + assert "web_search" in r2 + assert "call_1" in r2 # references the FIRST occurrence in the streak + assert len(r2) < len(_BIG) + + +def test_no_stub_when_fresh_result_differs(): + agent = _fake_agent() + args = {"query": "hermes"} + agent._append("web_search", args, _BIG, tool_call_id="call_1") + changed = "y" + _BIG + r2 = agent._append("web_search", args, changed, tool_call_id="call_2") + assert r2 == changed # changed result flows through whole + + +def test_changed_result_resets_streak_then_stub_references_new_first(): + agent = _fake_agent() + args = {"id": "job"} + agent._append("web_search", args, _BIG, tool_call_id="a") + changed = _BIG + "done" + r2 = agent._append("web_search", args, changed, tool_call_id="b") + assert r2 == changed + r3 = agent._append("web_search", args, changed, tool_call_id="c") + assert "byte-identical" in r3 + assert "tool_call_id b" in r3 # new streak's first occurrence, not 'a' + + +def test_no_stub_below_min_chars(): + agent = _fake_agent() + small = "x" * (IDENTICAL_RESULT_STUB_MIN_CHARS - 1) + args = {"query": "hermes"} + agent._append("web_search", args, small, tool_call_id="c1") + r2 = agent._append("web_search", args, small, tool_call_id="c2") + assert "byte-identical" not in r2 + assert r2.startswith(small) # full payload kept (pre-existing warning suffix allowed) + + +def test_no_stub_for_error_results(): + agent = _fake_agent() + err = "Error executing tool: " + _BIG + args = {"command": "boom"} + agent._append("terminal", args, err, failed=True, tool_call_id="c1") + r2 = agent._append("terminal", args, err, failed=True, tool_call_id="c2") + assert "byte-identical" not in r2 + assert r2.startswith(err) # models must see fresh errors whole + + +def test_pollers_get_stub_but_never_loop_notice(): + agent = _fake_agent() + args = {"id": "job1"} + results = [ + agent._append("bfl_flux3_get_result", args, _BIG, tool_call_id=f"c{i}") + for i in range(4) + ] + assert results[0] == _BIG + for r in results[1:]: + assert "byte-identical" in r # stubbed: unchanged poll saves context + assert "consecutive identical call" not in r # notice stays exempt + + +def test_third_identical_call_gets_stub_plus_loop_notice(): + agent = _fake_agent() + args = {"query": "hermes"} + agent._append("web_search", args, _BIG, tool_call_id="c1") + agent._append("web_search", args, _BIG, tool_call_id="c2") + r3 = agent._append("web_search", args, _BIG, tool_call_id="c3") + assert "byte-identical" in r3 # stub replaces the payload + assert "3rd consecutive identical call" in r3 # notice appended after it + assert r3.index("byte-identical") < r3.index("3rd consecutive") + + +def test_stub_carries_spillover_path_when_first_result_persisted(): + agent = _fake_agent() + args = {"query": "big"} + agent._tool_guardrails.record_persisted_result( + "c1", "/home/u/.hermes/cache/spillover/c1.txt" + ) + agent._append("web_search", args, _BIG, tool_call_id="c1") + r2 = agent._append("web_search", args, _BIG, tool_call_id="c2") + assert "/home/u/.hermes/cache/spillover/c1.txt" in r2 + + +def test_stub_includes_args_summary_for_compression_safety(): + agent = _fake_agent() + args = {"query": "hermes result stubbing", "limit": 5} + agent._append("web_search", args, _BIG, tool_call_id="c1") + r2 = agent._append("web_search", args, _BIG, tool_call_id="c2") + # Canonical-args preview so the model knows WHAT the call was even if + # the referenced message is later evicted by compression. + assert "hermes result stubbing" in r2 + + +def test_stub_args_summary_truncated_to_120_chars(): + c = ToolCallGuardrailController() + args = {"query": "q" * 500} + assert c.observe_call("web_search", args, _BIG, tool_call_id="c1").stub is None + stub = c.observe_call("web_search", args, _BIG, tool_call_id="c2").stub + assert stub is not None + args_part = stub.split("Args: ", 1)[1] + assert len(args_part) < 200 # ~120-char preview + ellipsis + closer + + +def test_config_off_disables_stub(): + agent = _fake_agent(stall_guards=False) + args = {"query": "hermes"} + agent._append("web_search", args, _BIG, tool_call_id="c1") + r2 = agent._append("web_search", args, _BIG, tool_call_id="c2") + assert "byte-identical" not in r2 + assert r2.startswith(_BIG) + + +def test_streak_reset_by_different_call_means_next_identical_is_full(): + agent = _fake_agent() + args = {"query": "hermes"} + agent._append("web_search", args, _BIG, tool_call_id="c1") + agent._append("read_file", {"path": "/a"}, "other", tool_call_id="c2") + r3 = agent._append("web_search", args, _BIG, tool_call_id="c3") + assert "byte-identical" not in r3 + assert r3.startswith(_BIG) # fresh streak — first occurrence full again + + +def test_multimodal_content_never_stubbed_and_breaks_streak(): + c = ToolCallGuardrailController() + args = {"path": "/img.png"} + assert c.observe_call("vision", args, _BIG, tool_call_id="c1").stub is None + # Non-string (multimodal) results never form or extend a streak. + obs = c.observe_call("vision", args, None, tool_call_id="c2") + assert obs.stub is None + assert c.observe_call("vision", args, _BIG, tool_call_id="c3").stub is None + + +def test_observe_identical_call_backcompat_notice_still_fires(): + c = ToolCallGuardrailController() + for _ in range(STALL_GUARD_IDENTICAL_CALL_THRESHOLD - 1): + assert c.observe_identical_call("web_search", {"q": 1}, "r") is None + assert c.observe_identical_call("web_search", {"q": 1}, "r") is not None + + +def test_extract_persisted_path_round_trip(): + # The stub's spillover reference is parsed from the + # block that maybe_persist_tool_result builds — assert the round trip. + from tools.tool_result_storage import ( + _build_persisted_message, + extract_persisted_path, + ) + + block = _build_persisted_message("preview", True, 50_000, "/tmp/spill/x.txt") + assert extract_persisted_path(block) == "/tmp/spill/x.txt" + assert extract_persisted_path("plain result") is None + + # ── said-continue-but-stopped detector ───────────────────────────────────── diff --git a/tools/tool_result_storage.py b/tools/tool_result_storage.py index ff731a5b8f..1fc3085677 100644 --- a/tools/tool_result_storage.py +++ b/tools/tool_result_storage.py @@ -295,6 +295,22 @@ def _build_persisted_message( return msg +_PERSISTED_PATH_RE = re.compile(r"^Full output saved to: (.+)$", re.MULTILINE) + + +def extract_persisted_path(content: str) -> str | None: + """Return the file path from a replacement block. + + Used by the result-reference stubbing guard (agent/tool_guardrails.py) so + a stub referencing a persisted first occurrence can carry the spillover + path instead of dangling. Returns None for non-persisted content. + """ + if not isinstance(content, str) or PERSISTED_OUTPUT_TAG not in content: + return None + match = _PERSISTED_PATH_RE.search(content) + return match.group(1).strip() if match else None + + def maybe_persist_tool_result( content: str, tool_name: str, diff --git a/website/docs/user-guide/configuration.md b/website/docs/user-guide/configuration.md index 28d58bd911..68b3adf19c 100644 --- a/website/docs/user-guide/configuration.md +++ b/website/docs/user-guide/configuration.md @@ -1752,6 +1752,8 @@ agent: stall_guards: false ``` +The same gate also enables **result-reference stubbing**: when a re-issued identical tool call returns a byte-identical fresh result, the duplicate payload enters context as a short reference stub pointing at the earlier result (tool name, `tool_call_id`, an args summary, and — if the first result was persisted to disk — its spillover path) instead of repeating the full output. The tool still executes every time, so polling semantics are preserved: a changed result always flows through whole. Results under 512 characters, error results, and multimodal results are never stubbed, and pollers *are* stubbed (an unchanged poll is exactly the case where the duplicate payload carries no information). + ## TTS Configuration ```yaml From c2f5d2da211fddf6841aa911a2d79406116203d8 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Thu, 20 Aug 2026 00:04:39 -0700 Subject: [PATCH 42/42] =?UTF-8?q?test:=20vary=20marathon-turn=20fixture=20?= =?UTF-8?q?args=20=E2=80=94=20identical=20calls=20now=20legitimately=20ded?= =?UTF-8?q?upe=20to=20stubs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tests/run_agent/test_compression_budget_refund.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/tests/run_agent/test_compression_budget_refund.py b/tests/run_agent/test_compression_budget_refund.py index b3d464d2e4..4235cd1c7d 100644 --- a/tests/run_agent/test_compression_budget_refund.py +++ b/tests/run_agent/test_compression_budget_refund.py @@ -72,7 +72,12 @@ def _tool_call(i: int): return SimpleNamespace( id=f"call_{i}", type="function", - function=SimpleNamespace(name="web_search", arguments='{"query": "x"}'), + # Vary the query per call: a real marathon turn issues distinct + # lookups, and identical (args, result) pairs are now legitimately + # deduped into reference stubs by the stall-guard subsystem — + # zero-variance args here would deflate the very context pressure + # this test exists to exercise. + function=SimpleNamespace(name="web_search", arguments=f'{{"query": "x{i}"}}'), )