Merge remote-tracking branch 'origin/main' into agent/81234-merge-20260821
This commit is contained in:
+17
-14
@@ -909,14 +909,12 @@ def init_agent(
|
||||
agent._active_children = [] # Running child AIAgents (for interrupt propagation)
|
||||
agent._active_children_lock = threading.Lock()
|
||||
|
||||
# Background memory/skill review state (agent/background_review.py). Holds
|
||||
# the forked review AIAgent while its run_conversation() is in flight, so
|
||||
# the NEXT live turn can proactively interrupt a still-running review
|
||||
# instead of letting the two race concurrently against the same
|
||||
# session_id/credentials (observed as doubled prompt-token counts and a
|
||||
# Ctrl+C-proof lockup when a live turn started before a review fired at
|
||||
# the end of the prior turn had finished).
|
||||
# Background memory/skill review state (agent/background_review.py).
|
||||
# ``_background_review_run`` is installed before the worker starts and
|
||||
# fences its first provider-capable phase; the direct agent pointer keeps
|
||||
# normal interrupt propagation available once the fork is constructed.
|
||||
agent._background_review_agent = None
|
||||
agent._background_review_run = None
|
||||
agent._background_review_lock = threading.Lock()
|
||||
|
||||
# Store OpenRouter provider preferences
|
||||
@@ -1815,13 +1813,18 @@ def init_agent(
|
||||
agent._memory_nudge_interval = 10
|
||||
agent._turns_since_memory = 0
|
||||
agent._iters_since_skill = 0
|
||||
# A flush/background agent may pass skip_memory=True to avoid spinning up an
|
||||
# external memory *provider*, but if the caller also explicitly enables the
|
||||
# "memory" toolset it still needs the built-in file-backed store — otherwise
|
||||
# the memory tool dispatches with store=None and every call fails (#65429).
|
||||
# So the built-in store is created unless memory is globally disabled, while
|
||||
# the external-provider block below stays gated on skip_memory.
|
||||
_memory_toolset_requested = "memory" in (agent.enabled_toolsets or [])
|
||||
# skip_memory=True skips the external memory *provider*. Flush/background
|
||||
# agents can still pass enabled_toolsets=["memory"] so the built-in file
|
||||
# store exists and the memory tool does not fail with store=None (#65429).
|
||||
# A toolset on disabled_toolsets is not a request: a caller that denylists
|
||||
# memory while its default toolset still names it must not get MEMORY.md
|
||||
# loaded by an enabled-only check. (Cron agents now run with
|
||||
# skip_memory=False and take the normal path here.)
|
||||
_enabled_toolsets = agent.enabled_toolsets or []
|
||||
_disabled_toolsets = agent.disabled_toolsets or []
|
||||
_memory_toolset_requested = (
|
||||
"memory" in _enabled_toolsets and "memory" not in _disabled_toolsets
|
||||
)
|
||||
if not skip_memory or _memory_toolset_requested:
|
||||
try:
|
||||
from tools.memory_tool import (
|
||||
|
||||
+205
-35
@@ -23,6 +23,7 @@ import json
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
import threading
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from agent.thread_scoped_output import thread_scoped_silence
|
||||
@@ -30,6 +31,158 @@ from agent.thread_scoped_output import thread_scoped_silence
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
_BACKGROUND_REVIEW_CANCEL_TIMEOUT_SECONDS = 2.0
|
||||
|
||||
|
||||
class _BackgroundReviewRun:
|
||||
"""Per-review cancellation and request-completion handshake."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.cancel_requested = threading.Event()
|
||||
self.request_done = threading.Event()
|
||||
self._lock = threading.Lock()
|
||||
self._review_agent = None
|
||||
self._request_finished = False
|
||||
self._cancel_dispatched = False
|
||||
|
||||
def begin_request(self, review_agent: Any) -> bool:
|
||||
"""Atomically admit the first provider-capable review phase."""
|
||||
with self._lock:
|
||||
if self.cancel_requested.is_set() or self._request_finished:
|
||||
return False
|
||||
self._review_agent = review_agent
|
||||
return True
|
||||
|
||||
def cancel(self) -> Any:
|
||||
"""Fence startup and return the running fork, if one was admitted."""
|
||||
with self._lock:
|
||||
self.cancel_requested.set()
|
||||
if self._review_agent is not None and not self._cancel_dispatched:
|
||||
self._cancel_dispatched = True
|
||||
return self._review_agent
|
||||
return None
|
||||
|
||||
def mark_request_finished(self) -> bool:
|
||||
"""Latch request completion once; the caller publishes the event."""
|
||||
with self._lock:
|
||||
if self._request_finished:
|
||||
return False
|
||||
self._request_finished = True
|
||||
self._review_agent = None
|
||||
return True
|
||||
|
||||
|
||||
def prepare_background_review_run(agent: Any) -> Optional[_BackgroundReviewRun]:
|
||||
"""Install a unique run token on the parent before ``Thread.start()``."""
|
||||
lock = getattr(agent, "_background_review_lock", None)
|
||||
if lock is None:
|
||||
try:
|
||||
lock = threading.Lock()
|
||||
agent._background_review_lock = lock
|
||||
except (AttributeError, TypeError):
|
||||
return None
|
||||
|
||||
run = _BackgroundReviewRun()
|
||||
try:
|
||||
with lock:
|
||||
current = getattr(agent, "_background_review_run", None)
|
||||
if current is not None and not current.request_done.is_set():
|
||||
return None
|
||||
agent._background_review_run = run
|
||||
except (AttributeError, TypeError):
|
||||
return None
|
||||
return run
|
||||
|
||||
|
||||
def finish_background_review_run(
|
||||
agent: Any,
|
||||
run: Optional[_BackgroundReviewRun],
|
||||
) -> None:
|
||||
"""Publish one run's request exit without clearing a successor (ABA-safe)."""
|
||||
if run is None or not run.mark_request_finished():
|
||||
return
|
||||
|
||||
lock = getattr(agent, "_background_review_lock", None)
|
||||
if lock is not None:
|
||||
with lock:
|
||||
if getattr(agent, "_background_review_run", None) is run:
|
||||
agent._background_review_run = None
|
||||
elif getattr(agent, "_background_review_run", None) is run:
|
||||
agent._background_review_run = None
|
||||
run.request_done.set()
|
||||
|
||||
|
||||
def _interrupt_background_review(review_agent: Any) -> None:
|
||||
"""Request abort off-thread so a broken abort hook cannot stall foreground.
|
||||
|
||||
The bounded wait on ``request_done`` in
|
||||
:func:`cancel_background_review_for_live_turn` is only effective if
|
||||
``interrupt()`` returns quickly. Off-loading to a daemon thread ensures
|
||||
a slow or wedged abort path cannot block the foreground turn (#84423).
|
||||
"""
|
||||
|
||||
def _interrupt() -> None:
|
||||
try:
|
||||
from agent.interrupt_compat import request_hard_interrupt
|
||||
|
||||
request_hard_interrupt(review_agent, "superseded by a new live turn")
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Failed to cancel in-flight background review for a new turn",
|
||||
exc_info=True,
|
||||
)
|
||||
|
||||
try:
|
||||
threading.Thread(
|
||||
target=_interrupt,
|
||||
daemon=True,
|
||||
name="bg-review-cancel",
|
||||
).start()
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Failed to start background-review cancellation thread",
|
||||
exc_info=True,
|
||||
)
|
||||
|
||||
|
||||
def cancel_background_review_for_live_turn(agent: Any) -> None:
|
||||
"""Cancel the current review and await its request-phase acknowledgement.
|
||||
|
||||
Foreground priority is preserved: if the review does not acknowledge within
|
||||
the bounded deadline, a warning is logged and the live turn proceeds
|
||||
anyway. The review is non-critical self-improvement work and must never
|
||||
block a user-facing turn (#84423).
|
||||
"""
|
||||
lock = getattr(agent, "_background_review_lock", None)
|
||||
if lock is not None:
|
||||
with lock:
|
||||
run = getattr(agent, "_background_review_run", None)
|
||||
legacy_agent = getattr(agent, "_background_review_agent", None)
|
||||
else:
|
||||
run = getattr(agent, "_background_review_run", None)
|
||||
legacy_agent = getattr(agent, "_background_review_agent", None)
|
||||
|
||||
if run is None:
|
||||
if legacy_agent is None:
|
||||
return
|
||||
_interrupt_background_review(legacy_agent)
|
||||
return
|
||||
|
||||
review_agent = run.cancel()
|
||||
if review_agent is not None:
|
||||
_interrupt_background_review(review_agent)
|
||||
|
||||
acknowledged = run.request_done.wait(
|
||||
timeout=_BACKGROUND_REVIEW_CANCEL_TIMEOUT_SECONDS
|
||||
)
|
||||
if not acknowledged:
|
||||
logger.warning(
|
||||
"Background review did not acknowledge cancellation within %.1fs; "
|
||||
"proceeding with foreground live turn",
|
||||
_BACKGROUND_REVIEW_CANCEL_TIMEOUT_SECONDS,
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Background-review aux-model selector + routed digest.
|
||||
#
|
||||
@@ -862,13 +1015,23 @@ def _run_review_in_thread(
|
||||
messages_snapshot: List[Dict],
|
||||
prompt: str,
|
||||
task_cfg: Optional[Dict[str, Any]] = None,
|
||||
review_run: Optional[_BackgroundReviewRun] = None,
|
||||
) -> None:
|
||||
"""Worker function executed in the background-review daemon thread.
|
||||
|
||||
Spawns a forked ``AIAgent`` inheriting the parent's runtime, runs the
|
||||
review prompt, and surfaces a compact action summary back to the user
|
||||
via ``agent._safe_print`` and ``agent.background_review_callback``.
|
||||
|
||||
``review_run`` is the per-review cancellation token from
|
||||
:func:`prepare_background_review_run`. If a live turn bumps the
|
||||
cancel generation before this review reaches its first provider call,
|
||||
the review aborts without entering ``run_conversation()`` (#84423).
|
||||
"""
|
||||
if review_run is not None and review_run.cancel_requested.is_set():
|
||||
finish_background_review_run(agent, review_run)
|
||||
return
|
||||
|
||||
# Local import to avoid a hard circular dep at module load.
|
||||
from run_agent import AIAgent
|
||||
from tools.terminal_tool import set_approval_callback as _set_approval_callback
|
||||
@@ -918,6 +1081,10 @@ def _run_review_in_thread(
|
||||
except (ValueError, AttributeError):
|
||||
pass
|
||||
|
||||
def _finish_request_phase(agent_ref) -> None:
|
||||
_unregister_review_agent(agent_ref)
|
||||
finish_background_review_run(agent, review_run)
|
||||
|
||||
try:
|
||||
# Silence stdout/stderr for THIS worker thread only. A process-global
|
||||
# ``contextlib.redirect_stdout(devnull)`` here would also blank
|
||||
@@ -1117,12 +1284,10 @@ def _run_review_in_thread(
|
||||
# Register this fork on the PARENT's _active_children (the same
|
||||
# list interrupt() fans out to for subagent delegation) and
|
||||
# _background_review_agent (a direct pointer the next live turn
|
||||
# uses to proactively cancel a still-running review). Without
|
||||
# this, a review still streaming when the next turn starts races
|
||||
# the live turn against the same session_id/credentials — producing
|
||||
# doubled prompt-token accounting and a Ctrl+C-proof lockup.
|
||||
# Best-effort: agents built without agent_init.py (test stubs)
|
||||
# degrade to "no cross-cancellation" rather than aborting the review.
|
||||
# uses to interrupt an admitted request). The per-review run token
|
||||
# separately fences startup and acknowledges request-phase exit.
|
||||
# The legacy pointer/list remain best-effort for direct test stubs;
|
||||
# a prepared run token is the live-turn cancellation authority.
|
||||
if hasattr(agent, "_background_review_agent"):
|
||||
_br_lock = getattr(agent, "_background_review_lock", None)
|
||||
if _br_lock is not None:
|
||||
@@ -1173,22 +1338,26 @@ def _run_review_in_thread(
|
||||
pass
|
||||
|
||||
try:
|
||||
# Routed to a different model -> replay a digest (cache is cold
|
||||
# on that model anyway, so minimise cold-written tokens). Same
|
||||
# model -> replay the full snapshot (warm cache reads).
|
||||
_review_history = (
|
||||
_digest_history(messages_snapshot) if _routed
|
||||
else messages_snapshot
|
||||
)
|
||||
review_agent.run_conversation(
|
||||
user_message=(
|
||||
prompt
|
||||
+ "\n\nYou can only call memory and skill "
|
||||
"management tools. Other tools will be denied "
|
||||
"at runtime — do not attempt them."
|
||||
),
|
||||
conversation_history=_review_history,
|
||||
request_admitted = (
|
||||
review_run is None or review_run.begin_request(review_agent)
|
||||
)
|
||||
if request_admitted:
|
||||
# Routed to a different model -> replay a digest (cache is cold
|
||||
# on that model anyway, so minimise cold-written tokens). Same
|
||||
# model -> replay the full snapshot (warm cache reads).
|
||||
_review_history = (
|
||||
_digest_history(messages_snapshot) if _routed
|
||||
else messages_snapshot
|
||||
)
|
||||
review_agent.run_conversation(
|
||||
user_message=(
|
||||
prompt
|
||||
+ "\n\nYou can only call memory and skill "
|
||||
"management tools. Other tools will be denied "
|
||||
"at runtime — do not attempt them."
|
||||
),
|
||||
conversation_history=_review_history,
|
||||
)
|
||||
finally:
|
||||
clear_thread_tool_whitelist()
|
||||
# Attribute the review fork's usage to the PARENT session.
|
||||
@@ -1199,12 +1368,9 @@ def _run_review_in_thread(
|
||||
if review_agent is not None:
|
||||
review_usage.update(_snapshot_review_usage(review_agent))
|
||||
_record_review_usage_to_parent(agent, review_usage)
|
||||
# Unregister as soon as run_conversation() itself has
|
||||
# returned — that's the only phase making outbound API
|
||||
# calls, i.e. the only phase that can race the parent's
|
||||
# next live turn. Runs on both the success and exception
|
||||
# path (this whole block is inside the try/finally above).
|
||||
_unregister_review_agent(review_agent)
|
||||
# Publish completion as soon as the provider-capable phase has
|
||||
# returned or startup cancellation has fenced it out.
|
||||
_finish_request_phase(review_agent)
|
||||
|
||||
# Snapshot review actions before teardown. close() is allowed to
|
||||
# clean per-session state, but the user-visible self-improvement
|
||||
@@ -1284,13 +1450,10 @@ def _run_review_in_thread(
|
||||
# thread-scoped silence here so teardown output (Honcho flush, Hindsight
|
||||
# sync, background thread joins) stays quiet even on the exception path,
|
||||
# without blanking other threads' streams.
|
||||
# Also a safety-net unregister: covers exceptions raised during setup
|
||||
# (between registration and the run_conversation try/finally above)
|
||||
# that the primary _unregister_review_agent call site never reaches.
|
||||
# _unregister_review_agent is idempotent (checks `is`/`in` membership),
|
||||
# so calling it again here after the primary call site already ran is
|
||||
# a harmless no-op.
|
||||
_unregister_review_agent(review_agent)
|
||||
# Also a safety-net completion: covers exceptions raised during setup
|
||||
# before the request-phase finally. Both tracking cleanup and the
|
||||
# per-run completion publication are identity-scoped and idempotent.
|
||||
_finish_request_phase(review_agent)
|
||||
if review_agent is not None:
|
||||
try:
|
||||
with thread_scoped_silence():
|
||||
@@ -1319,6 +1482,7 @@ def spawn_background_review_thread(
|
||||
review_skills: bool = False,
|
||||
focus: Optional[str] = None,
|
||||
task_cfg: Optional[Dict[str, Any]] = None,
|
||||
review_run: Optional[_BackgroundReviewRun] = None,
|
||||
):
|
||||
"""Build the review thread target and prompt for a background review.
|
||||
|
||||
@@ -1359,7 +1523,13 @@ def spawn_background_review_thread(
|
||||
)
|
||||
|
||||
def _target() -> None:
|
||||
_run_review_in_thread(agent, messages_snapshot, prompt, task_cfg)
|
||||
_run_review_in_thread(
|
||||
agent,
|
||||
messages_snapshot,
|
||||
prompt,
|
||||
task_cfg=task_cfg,
|
||||
review_run=review_run,
|
||||
)
|
||||
|
||||
return _target, prompt
|
||||
|
||||
|
||||
@@ -479,6 +479,12 @@ def _chat_messages_to_responses_input(
|
||||
conversation is still on the wire.
|
||||
"""
|
||||
items: List[Dict[str, Any]] = []
|
||||
# Parallel to `items`: the raw chat message each converted item came
|
||||
# from. Pruning needs this to read a canonical summary carrier's
|
||||
# up-to-date, provenance-tagged content directly — the converted `item`
|
||||
# can be a lossy shape (stale exact-replay, or a typed
|
||||
# `function_call_output` wrapper) that no longer carries it (#90976).
|
||||
item_sources: List[Optional[Dict[str, Any]]] = []
|
||||
seen_item_ids: set = set()
|
||||
|
||||
for msg in messages:
|
||||
@@ -567,6 +573,7 @@ def _chat_messages_to_responses_input(
|
||||
if k not in ("id", "_issuer_kind")
|
||||
}
|
||||
items.append(replay_item)
|
||||
item_sources.append(msg)
|
||||
if item_id:
|
||||
seen_item_ids.add(item_id)
|
||||
has_codex_reasoning = True
|
||||
@@ -623,14 +630,17 @@ def _chat_messages_to_responses_input(
|
||||
if isinstance(phase, str) and phase.strip():
|
||||
replay_item["phase"] = phase.strip()
|
||||
items.append(replay_item)
|
||||
item_sources.append(msg)
|
||||
replayed_message_items += 1
|
||||
|
||||
if replayed_message_items > 0:
|
||||
pass
|
||||
elif content_parts:
|
||||
items.append({"role": "assistant", "content": content_parts})
|
||||
item_sources.append(msg)
|
||||
elif content_text.strip():
|
||||
items.append({"role": "assistant", "content": content_text})
|
||||
item_sources.append(msg)
|
||||
elif has_codex_reasoning:
|
||||
# The Responses API requires a following item after each
|
||||
# reasoning item (otherwise: missing_following_item error).
|
||||
@@ -638,6 +648,7 @@ def _chat_messages_to_responses_input(
|
||||
# content, emit an empty assistant message as the required
|
||||
# following item.
|
||||
items.append({"role": "assistant", "content": ""})
|
||||
item_sources.append(msg)
|
||||
|
||||
tool_calls = msg.get("tool_calls")
|
||||
if isinstance(tool_calls, list):
|
||||
@@ -680,6 +691,7 @@ def _chat_messages_to_responses_input(
|
||||
"name": fn_name,
|
||||
"arguments": arguments,
|
||||
})
|
||||
item_sources.append(msg)
|
||||
continue
|
||||
|
||||
# Non-assistant (user) role: emit multimodal parts when present,
|
||||
@@ -688,6 +700,7 @@ def _chat_messages_to_responses_input(
|
||||
items.append({"role": role, "content": content_parts})
|
||||
else:
|
||||
items.append({"role": role, "content": content_text})
|
||||
item_sources.append(msg)
|
||||
continue
|
||||
|
||||
if role == "tool":
|
||||
@@ -722,24 +735,38 @@ def _chat_messages_to_responses_input(
|
||||
"call_id": _clamp_responses_call_id(call_id),
|
||||
"output": output_value,
|
||||
})
|
||||
item_sources.append(msg)
|
||||
|
||||
# Native server-side compaction: when a replayed checkpoint is present,
|
||||
# restructure the wire around it. The server renders nothing placed
|
||||
# before a compaction item (live-verified Aug 2026), so pre-checkpoint
|
||||
# history is dead upload weight and — worse — the user's plaintext asks
|
||||
# from before the boundary silently vanish from the model's view. Keep
|
||||
# the newest checkpoint first, retain pre-checkpoint USER messages
|
||||
# verbatim within a token budget (Codex CLI parity), and leave the
|
||||
# history is dead upload weight and — worse — the user's plaintext asks,
|
||||
# and any local-compression summary already merged into that history,
|
||||
# silently vanish from the model's view. Keep the newest checkpoint
|
||||
# first, retain pre-checkpoint USER messages and compression-SUMMARY
|
||||
# messages (whole, never byte-sliced) verbatim within a token budget
|
||||
# each (Codex CLI parity for the user side), and leave the
|
||||
# post-checkpoint tail untouched. Gated on the CURRENT request's native
|
||||
# eligibility, not merely on the presence of a checkpoint: a persisted
|
||||
# checkpoint outlives the gate, and pruning for a request that carries no
|
||||
# ``context_management`` deletes history the server never compacted.
|
||||
#
|
||||
# ``item_sources`` (parallel to ``items``) carries the raw chat message
|
||||
# each converted item came from. A canonical summary carrier's content
|
||||
# can be lost or gone stale by the time it becomes a Responses item — a
|
||||
# merge-into-tail tool-result carrier becomes a typed
|
||||
# ``function_call_output`` (no ``content``/``role`` at all), and a
|
||||
# merge-into-tail assistant carrier can be shadowed by a stale exact
|
||||
# ``codex_message_items`` replay from before the merge rewrote its
|
||||
# content. Pruning reads the source message's own up-to-date,
|
||||
# provenance-tagged content directly instead of trying to recover it
|
||||
# from whatever shape the conversion produced (#90976).
|
||||
if not native_compaction_eligible:
|
||||
return items
|
||||
|
||||
from agent.native_compaction import prune_pre_checkpoint_items
|
||||
|
||||
return prune_pre_checkpoint_items(items)
|
||||
return prune_pre_checkpoint_items(items, item_sources=item_sources)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -1820,31 +1820,6 @@ def run_conversation(
|
||||
agent._last_compression_attempt_recorded = False
|
||||
agent._last_compression_attempt_in_place = None
|
||||
|
||||
# If a background memory/skill review spawned at the end of a PRIOR turn
|
||||
# (agent/background_review.py) is still running its own run_conversation()
|
||||
# when THIS turn starts, cancel it now rather than letting both make
|
||||
# outbound API calls concurrently against the same session_id/credentials.
|
||||
# That concurrency can produce doubled prompt-token accounting on this
|
||||
# turn's own calls and, because the review fork is a fully separate
|
||||
# AIAgent with no route back to THIS agent's interrupt() by default, a
|
||||
# lockup that survives a normal /stop and needs a hard Ctrl+C.
|
||||
# ``review_agent.interrupt()`` is fire-and-forget here — it just flags
|
||||
# cancellation and aborts the review's in-flight socket; it does not
|
||||
# block waiting for the review's daemon thread to exit, so it can't add
|
||||
# latency to this turn. Only ever set on the real owning agent (the
|
||||
# review fork's own copy of this attribute stays None — reviews don't
|
||||
# spawn nested reviews), so this is a no-op on every other run_conversation
|
||||
# caller (subagents, the review fork itself, etc).
|
||||
_pending_review = getattr(agent, "_background_review_agent", None)
|
||||
if _pending_review is not None:
|
||||
try:
|
||||
_pending_review.interrupt("superseded by a new live turn")
|
||||
except Exception:
|
||||
logger.debug(
|
||||
"Failed to cancel in-flight background review for a new turn",
|
||||
exc_info=True,
|
||||
)
|
||||
|
||||
# Adopt any ~/.hermes/.env credential/base-url edits made since the last
|
||||
# turn — a Settings save updates .env but not this worker's client, which
|
||||
# was built at agent init (#67821). No-op when .env is unchanged.
|
||||
|
||||
@@ -508,6 +508,9 @@ DEFAULT_CONTEXT_LENGTHS = {
|
||||
# ensures "glm-5.2" resolves to 1M while older variants still hit the
|
||||
# generic 202K fallback.
|
||||
"glm-5.2": 1_048_576,
|
||||
# OpenRouter's free GLM-5.2 variant is capped at 256K (live metadata,
|
||||
# 2026-08-21) — longer key wins over the 1M paid entry above.
|
||||
"glm-5.2:free": 256_000,
|
||||
"glm": 202752,
|
||||
# xAI Grok — xAI /v1/models does not return context_length metadata,
|
||||
# so these hardcoded fallbacks prevent Hermes from probing-down to
|
||||
@@ -563,7 +566,15 @@ DEFAULT_CONTEXT_LENGTHS = {
|
||||
# (stealth/ox-alpha). 1M context per OpenRouter live metadata (2026-08-20).
|
||||
"ox-alpha": 1_048_576,
|
||||
# Nemotron — NVIDIA's open-weights series (128K context across all sizes)
|
||||
# EXCEPT 3.5 Lightning, which ships a 1M window (OpenRouter live metadata
|
||||
# + OpenCode Zen free tier, verified 2026-08-21).
|
||||
"nemotron-3.5-lightning": 1_000_000,
|
||||
"nemotron": 131072,
|
||||
# Poolside Laguna 2.1 (s/xs) — 256K window per OpenRouter live metadata
|
||||
# (2026-08-21). Covers laguna-s-2.1:free, laguna-xs-2.1:free, and the
|
||||
# OpenCode Zen laguna-s-2.1-free slug via substring matching.
|
||||
"laguna-s-2.1": 262144,
|
||||
"laguna-xs-2.1": 262144,
|
||||
# Arcee
|
||||
"trinity": 262144,
|
||||
# OpenRouter
|
||||
|
||||
+199
-63
@@ -32,15 +32,25 @@ captured compaction items ride the existing ``codex_reasoning_items``
|
||||
sidecar, which already handles persistence (state.db), gateway session
|
||||
replay, cross-issuer stamping, and the encrypted-replay kill switch.
|
||||
|
||||
This module is dependency-free on purpose so the transport, adapter, and
|
||||
conversation loop can share the gate without import cycles.
|
||||
This module stays free of transport/adapter dependencies so the transport,
|
||||
adapter, and conversation loop can share the gate without import cycles. The
|
||||
two exceptions — ``agent.context_compressor`` and ``agent.message_content`` —
|
||||
sit below this module in the dependency graph (neither imports
|
||||
``native_compaction``), so importing their provenance/text primitives here
|
||||
introduces no cycle.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
from urllib.parse import urlsplit
|
||||
|
||||
from agent.context_compressor import is_compaction_summary_message
|
||||
from agent.message_content import flatten_message_text
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Native compaction fires this many tokens below the local compressor's
|
||||
# trigger so the server always gets the first shot at compaction.
|
||||
LOCAL_TRIGGER_SAFETY_MARGIN = 8_192
|
||||
@@ -147,73 +157,130 @@ def native_compaction_context_management(
|
||||
# Retention budget for plaintext user messages carried across a native
|
||||
# compaction boundary (mirrors Codex CLI's RETAINED_MESSAGE_TOKEN_BUDGET).
|
||||
# Live verification (Aug 2026, gpt-5.6 @ api.openai.com): the server renders
|
||||
# NOTHING placed before a replayed compaction checkpoint — a fact stated in a
|
||||
# pre-checkpoint input item is invisible to the model ("NONE" recall), while
|
||||
# the same item placed after the checkpoint recalls perfectly. Without
|
||||
# retention, every plaintext user ask from before the compaction survives
|
||||
# only as whatever the opaque server summary kept — the goal-drift failure
|
||||
# mode. Codex CLI solves this by rebuilding history with user messages
|
||||
# retained verbatim; ``prune_pre_checkpoint_items`` is our wire-level
|
||||
# equivalent.
|
||||
RETAINED_USER_MESSAGE_TOKEN_BUDGET = 64_000
|
||||
|
||||
# Retention budget for local compression summary messages carried across a native
|
||||
# compaction boundary to prevent summary token inflation.
|
||||
RETAINED_SUMMARY_TOKEN_BUDGET = 32_000
|
||||
|
||||
|
||||
def _approx_tokens(text: str) -> int:
|
||||
"""Cheap chars//4 token estimate — same shape Codex uses for retention."""
|
||||
return max(1, len(text) // 4)
|
||||
|
||||
|
||||
def _user_item_text(item: Dict[str, Any]) -> Optional[str]:
|
||||
"""Extract the retained-budget text of a user-role input item.
|
||||
def _extract_item_text(item: Any) -> Optional[str]:
|
||||
"""Extract measurable text from string, list content, output_text, or nested metadata text.
|
||||
|
||||
Returns None when the item carries no measurable text (empty message).
|
||||
Multimodal list content is measured by its ``input_text`` parts; images
|
||||
count as zero, matching Codex's retention accounting.
|
||||
Returns None when the item carries no measurable text.
|
||||
Handles string content, multipart lists (input_text/text/output_text), and fallback keys.
|
||||
"""
|
||||
if not isinstance(item, dict):
|
||||
return None
|
||||
|
||||
content = item.get("content")
|
||||
if content is None and "output_text" in item:
|
||||
content = item.get("output_text")
|
||||
|
||||
if isinstance(content, str):
|
||||
return content if content.strip() else None
|
||||
|
||||
if isinstance(content, list):
|
||||
text = "".join(
|
||||
part.get("text", "")
|
||||
for part in content
|
||||
if isinstance(part, dict) and part.get("type") == "input_text"
|
||||
)
|
||||
return text if text.strip() or content else None
|
||||
parts = []
|
||||
for part in content:
|
||||
if isinstance(part, str):
|
||||
if part.strip():
|
||||
parts.append(part.strip())
|
||||
elif isinstance(part, dict):
|
||||
part_text = part.get("text") or part.get("input_text") or part.get("output_text")
|
||||
if isinstance(part_text, str) and part_text.strip():
|
||||
parts.append(part_text.strip())
|
||||
part_meta = part.get("metadata")
|
||||
if isinstance(part_meta, dict) and isinstance(part_meta.get("text"), str):
|
||||
if part_meta["text"].strip():
|
||||
parts.append(part_meta["text"].strip())
|
||||
text = " ".join(parts)
|
||||
return text if text.strip() else None
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _is_summary_item(item: Any) -> bool:
|
||||
"""True when *item* is a canonical Hermes compression-summary message.
|
||||
|
||||
Delegates entirely to
|
||||
``agent.context_compressor.is_compaction_summary_message`` — the single
|
||||
authoritative provenance check already used by every other summary
|
||||
consumer (memory providers, frontends, the compactor itself). It prefers
|
||||
the exact, truthy ``COMPRESSED_SUMMARY_METADATA_KEY`` marker and falls
|
||||
back to the canonical prefix classifier (``SUMMARY_PREFIX`` /
|
||||
``LEGACY_SUMMARY_PREFIX`` / historical prefixes, including the
|
||||
merge-into-tail shape) for the case where the underscore-prefixed key
|
||||
was already stripped by a wire sanitizer.
|
||||
|
||||
Deliberately NOT a second heuristic: no arbitrary underscore-key scan, no
|
||||
inference from a falsy or unrelated metadata key, and no matching on
|
||||
ad-hoc content headings like ``"## Summary"`` in ordinary text — any of
|
||||
those can promote a normal user/assistant message (or adversarial
|
||||
content) to durable retained history (#90975 review).
|
||||
"""
|
||||
return is_compaction_summary_message(item)
|
||||
|
||||
|
||||
def prune_pre_checkpoint_items(
|
||||
items: List[Dict[str, Any]],
|
||||
retained_user_token_budget: int = RETAINED_USER_MESSAGE_TOKEN_BUDGET,
|
||||
retained_summary_token_budget: int = RETAINED_SUMMARY_TOKEN_BUDGET,
|
||||
enable_summary_retention: bool = True,
|
||||
item_sources: Optional[List[Any]] = None,
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Restructure Responses input around the newest compaction checkpoint.
|
||||
|
||||
The server drops every input item that precedes a replayed ``compaction``
|
||||
item (live-verified Aug 2026), so sending pre-checkpoint history is dead
|
||||
weight AND silently erases the user's plaintext asks. When a checkpoint
|
||||
is present, rebuild the wire as::
|
||||
weight AND silently erases the user's plaintext asks — including any
|
||||
local-compression summary the agent already produced, which previously
|
||||
vanished here because it carries ``role="assistant"``, not ``"user"``
|
||||
(#90975). When a checkpoint is present, rebuild the wire as::
|
||||
|
||||
[checkpoint run] + [retained user messages (newest-first budget)] + [post]
|
||||
[checkpoint run] + [retained user & summary messages (newest-first budget)] + [post]
|
||||
|
||||
- The NEWEST contiguous run of checkpoints wins (the server can emit
|
||||
more than one compaction item in a single response — live-observed
|
||||
Aug 2026 — and they arrive adjacent; a run from a newer response
|
||||
cumulatively carries prior windows, so older runs are dropped).
|
||||
- Retained user messages are the user-role items from before the
|
||||
checkpoint, kept verbatim newest-first within
|
||||
- The NEWEST contiguous run of checkpoints wins.
|
||||
- Retained user messages are kept verbatim within
|
||||
``retained_user_token_budget``; the boundary message is head-truncated
|
||||
when it only partially fits (string content only).
|
||||
- Everything after the checkpoint is untouched, so function_call /
|
||||
function_call_output pairing is preserved (a checkpoint is captured on
|
||||
an assistant response, and that response's own calls and their outputs
|
||||
are all emitted after its reasoning items).
|
||||
- No checkpoint in ``items`` → returned unchanged (self-gating: non-native
|
||||
routes and kill-switched sessions never see a restructured wire).
|
||||
|
||||
Deterministic for a given history, so the request prefix stays stable
|
||||
across turns and server-side prompt caching keeps working.
|
||||
when it only partially fits (string content only) — goals are usually
|
||||
stated up front, so the head is the valuable end.
|
||||
- Compression summary messages (``_is_summary_item``, the canonical
|
||||
``agent.context_compressor`` provenance check) are retained whole
|
||||
within ``retained_summary_token_budget``. A summary is never
|
||||
byte/character-sliced: Hermes summaries carry structural framing
|
||||
(handoff prefix, end marker, merge-into-tail delimiters) that a blind
|
||||
slice can corrupt, so one that doesn't fit whole is dropped instead.
|
||||
A summary already retained once (identical text) is never duplicated,
|
||||
so repeated checkpoints stay idempotent.
|
||||
- ``enable_summary_retention`` is a function-level override (used by
|
||||
tests and callers that need the pre-#90975 behavior back); it is not
|
||||
wired to a user-facing config surface.
|
||||
- Original relative chronological order between user messages and
|
||||
summaries is preserved.
|
||||
- ``item_sources`` (optional, parallel to ``items``) is the raw chat
|
||||
message each Responses item was converted from. By the time a summary
|
||||
reaches this function as a converted ``item`` it can already be lossy:
|
||||
a merge-into-tail tool-result carrier becomes a typed
|
||||
``function_call_output`` (no ``content``/``role`` survives the
|
||||
conversion at all), and a merge-into-tail assistant carrier can be
|
||||
shadowed by a stale exact ``codex_message_items`` replay captured
|
||||
before the merge rewrote its content. When a source is provided and is
|
||||
itself a canonical summary carrier (``is_compaction_summary_message``),
|
||||
its content is read directly from the source — never from the
|
||||
converted item — and it is retained as a synthesized
|
||||
``role="assistant"`` message regardless of what shape the original
|
||||
item took. Without ``item_sources`` (default), retention only sees
|
||||
what survived conversion, matching pre-#90976 behavior (#90976).
|
||||
"""
|
||||
if not isinstance(items, list) or not items:
|
||||
return items
|
||||
|
||||
last_cp = None
|
||||
for i, item in enumerate(items):
|
||||
if isinstance(item, dict) and item.get("type") == "compaction":
|
||||
@@ -234,36 +301,105 @@ def prune_pre_checkpoint_items(
|
||||
checkpoint_run = items[first_cp : last_cp + 1]
|
||||
post = items[last_cp + 1 :]
|
||||
|
||||
if isinstance(item_sources, list) and len(item_sources) == len(items):
|
||||
pre_sources: List[Any] = item_sources[:first_cp]
|
||||
else:
|
||||
pre_sources = [None] * len(pre)
|
||||
|
||||
retained_reversed: List[Dict[str, Any]] = []
|
||||
remaining = max(0, int(retained_user_token_budget))
|
||||
for item in reversed(pre):
|
||||
if not isinstance(item, dict) or item.get("role") != "user":
|
||||
user_remaining = max(0, int(retained_user_token_budget))
|
||||
summary_remaining = max(0, int(retained_summary_token_budget))
|
||||
seen_summary_texts: set = set()
|
||||
|
||||
def _try_retain_summary(text: Optional[str]) -> Optional[Dict[str, Any]]:
|
||||
"""Check budget/dedup/cost for a summary; return cost info or None."""
|
||||
if not text or summary_remaining <= 0 or text in seen_summary_texts:
|
||||
return None
|
||||
cost = _approx_tokens(text)
|
||||
if cost > summary_remaining:
|
||||
# Never byte-slice a summary's structural framing — drop it
|
||||
# whole rather than corrupt the handoff prefix / end marker.
|
||||
return None
|
||||
seen_summary_texts.add(text)
|
||||
return {"cost": cost}
|
||||
|
||||
for item, source in zip(reversed(pre), reversed(pre_sources)):
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
# Skip typed items (function_call_output etc. never carry role=user,
|
||||
# but stay defensive about future shapes).
|
||||
|
||||
# Canonical source-based summary detection: reads the ORIGINAL chat
|
||||
# message's own content, so it sees past a lossy conversion (a
|
||||
# typed `function_call_output` wrapper, or a stale exact-replay
|
||||
# message) that erased the summary from `item` itself (#90976).
|
||||
# This is never a heuristic promotion of arbitrary item content —
|
||||
# it only fires when the source message itself is a canonical,
|
||||
# provenance-tagged summary carrier.
|
||||
if enable_summary_retention and isinstance(source, dict) and _is_summary_item(source):
|
||||
text = flatten_message_text(source.get("content")) if isinstance(source, dict) else ""
|
||||
text = text if text.strip() else None
|
||||
result = _try_retain_summary(text)
|
||||
if result:
|
||||
_src_role = source.get("role")
|
||||
retained_reversed.append({
|
||||
"role": _src_role if _src_role in ("user", "assistant") else "assistant",
|
||||
"content": text,
|
||||
})
|
||||
summary_remaining -= result["cost"]
|
||||
continue
|
||||
|
||||
# Skip typed non-message items (function_call_output etc. never
|
||||
# carry role=user or a summary flag, but stay defensive about
|
||||
# future shapes).
|
||||
if "type" in item and item.get("type") != "message":
|
||||
continue
|
||||
if remaining <= 0:
|
||||
break
|
||||
text = _user_item_text(item)
|
||||
|
||||
is_summary = enable_summary_retention and _is_summary_item(item)
|
||||
is_user = item.get("role") == "user"
|
||||
|
||||
if not is_user and not is_summary:
|
||||
continue
|
||||
|
||||
text = _extract_item_text(item)
|
||||
if text is None:
|
||||
continue
|
||||
cost = _approx_tokens(text)
|
||||
if cost <= remaining:
|
||||
retained_reversed.append(item)
|
||||
remaining -= cost
|
||||
elif isinstance(item.get("content"), str):
|
||||
# Head-truncate the boundary message: goals are usually stated
|
||||
# up front, so the head is the valuable end.
|
||||
truncated = dict(item)
|
||||
truncated["content"] = item["content"][: remaining * 4]
|
||||
if truncated["content"].strip():
|
||||
retained_reversed.append(truncated)
|
||||
remaining = 0
|
||||
# Multimodal boundary message that doesn't fit whole: skip rather
|
||||
# than rewrite parts.
|
||||
# Image-only user messages have empty text but non-empty content —
|
||||
# main retains them at 1-token cost (images count as zero, matching
|
||||
# Codex's retention accounting). Don't skip them just because text
|
||||
# is falsy.
|
||||
if not text and not is_user:
|
||||
continue
|
||||
|
||||
return checkpoint_run + list(reversed(retained_reversed)) + post
|
||||
if is_summary:
|
||||
result = _try_retain_summary(text)
|
||||
if result:
|
||||
retained_reversed.append(item)
|
||||
summary_remaining -= result["cost"]
|
||||
elif is_user:
|
||||
if user_remaining <= 0:
|
||||
continue
|
||||
cost = _approx_tokens(text)
|
||||
if cost <= user_remaining:
|
||||
retained_reversed.append(item)
|
||||
user_remaining -= cost
|
||||
elif isinstance(item.get("content"), str):
|
||||
truncated = dict(item)
|
||||
truncated["content"] = item["content"][: user_remaining * 4]
|
||||
if truncated["content"].strip():
|
||||
retained_reversed.append(truncated)
|
||||
user_remaining = 0
|
||||
|
||||
retained_ordered = list(reversed(retained_reversed))
|
||||
result = checkpoint_run + retained_ordered + post
|
||||
|
||||
logger.debug(
|
||||
"Pruned pre-checkpoint items: %d input -> %d retained (user_rem=%d, summary_rem=%d)",
|
||||
len(items),
|
||||
len(result),
|
||||
user_remaining,
|
||||
summary_remaining,
|
||||
)
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def is_native_compaction_rejection(error: Any, status_code: Any = None) -> bool:
|
||||
|
||||
@@ -95,6 +95,13 @@ KIMI_K3_EFFORTS: tuple[str, ...] = ("low", "high", "max")
|
||||
#: Moonshot/Kimi K2-era models: low/medium/high.
|
||||
KIMI_K2_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
|
||||
#: OpenCode "Ox Alpha" stealth model (x-preview-f-free): thinking is always
|
||||
#: on and the wire accepts exactly low/high/max — medium/none/xhigh 400 with
|
||||
#: "This model always engages in thinking and cannot be disabled; please use
|
||||
#: low, high, or max" (verified live 2026-08-21). xhigh rounds up to max.
|
||||
OX_ALPHA_EFFORTS: tuple[str, ...] = ("low", "high", "max")
|
||||
OX_ALPHA_OVERRIDES: dict[str, str] = {"xhigh": "max"}
|
||||
|
||||
#: Tencent TokenHub: low/medium/high.
|
||||
TOKENHUB_EFFORTS: tuple[str, ...] = ("low", "medium", "high")
|
||||
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
/**
|
||||
* E2E contract for the compositor-only GlyphSpinner.
|
||||
*
|
||||
* The spinner's whole reason for existing in this shape is a CSS animation:
|
||||
* every frame is in the DOM from mount and a `transform` keyframes animation
|
||||
* scrolls between them, so there is no JS timer and no per-tick DOM mutation
|
||||
* scheduling document-scale style recalculation.
|
||||
*
|
||||
* None of that is observable in jsdom — it has no animation engine, no
|
||||
* cascade resolution for `steps()`, and no `Element.getAnimations()`. The
|
||||
* jsdom suite (src/components/ui/glyph-spinner.test.tsx) therefore pins the
|
||||
* DATA and WIRING, and this spec pins the RENDERED BEHAVIOUR in a real
|
||||
* browser, which is the only place the stylesheet actually runs.
|
||||
*
|
||||
* This replaces three tests that asserted on the TEXT of the stylesheet.
|
||||
* Reading source in a test is banned outright (AGENTS.md) and those tests
|
||||
* proved the point: a var()-fallback edit that changed no rendered pixel
|
||||
* broke one of them, while none of them had ever executed the CSS.
|
||||
*
|
||||
* Prerequisite: `npm run build` must have been run so dist/ exists.
|
||||
*/
|
||||
|
||||
import { expect, type Page, test } from '@playwright/test'
|
||||
|
||||
import { type MockBackendFixture, setupMockBackend, waitForAppReady } from './fixtures'
|
||||
|
||||
const STRIP = '.glyph-spinner__strip'
|
||||
|
||||
/**
|
||||
* Send a message so a turn is in flight — the composer status stack mounts a
|
||||
* GlyphSpinner while the agent is working. Resolves once a frame strip is in
|
||||
* the DOM.
|
||||
*/
|
||||
async function mountSpinner(page: Page): Promise<void> {
|
||||
const composer = page.locator('[contenteditable="true"]').first()
|
||||
await composer.waitFor({ state: 'visible', timeout: 10_000 })
|
||||
await composer.click()
|
||||
await composer.type('hello from the glyph spinner spec', { delay: 10 })
|
||||
await page.keyboard.press('Enter')
|
||||
|
||||
await page.waitForSelector(STRIP, { state: 'attached', timeout: 20_000 })
|
||||
}
|
||||
|
||||
test.describe('GlyphSpinner (compositor animation)', () => {
|
||||
let fixture: MockBackendFixture
|
||||
|
||||
test.beforeAll(async () => {
|
||||
fixture = await setupMockBackend()
|
||||
await waitForAppReady(fixture)
|
||||
})
|
||||
|
||||
test.afterAll(async () => {
|
||||
await fixture?.cleanup()
|
||||
})
|
||||
|
||||
test('animates with a steps() transform keyframes animation, one step per frame', async () => {
|
||||
const { page } = fixture
|
||||
await mountSpinner(page)
|
||||
|
||||
const observed = await page.evaluate(strip => {
|
||||
const el = document.querySelector<HTMLElement>(strip)
|
||||
|
||||
if (!el) {
|
||||
throw new Error('no frame strip in the DOM')
|
||||
}
|
||||
|
||||
const style = getComputedStyle(el)
|
||||
const animations = el.getAnimations()
|
||||
|
||||
return {
|
||||
frameCount: el.querySelectorAll('.glyph-spinner__frame').length,
|
||||
timingFunction: style.animationTimingFunction,
|
||||
iterationCount: style.animationIterationCount,
|
||||
durationMs: animations[0]?.effect?.getTiming().duration ?? null,
|
||||
names: animations.map(a => (a as CSSAnimation).animationName),
|
||||
// A percentage translate makes the animation layout-dependent, which
|
||||
// Chromium refuses to composite. Read the engine's own keyframes: a
|
||||
// revert to translateY(-100%) shows up here, while the computed
|
||||
// `style.transform` always serializes to a matrix and can't tell.
|
||||
travel: ((animations[0]?.effect as KeyframeEffect | undefined)?.getKeyframes() ?? [])
|
||||
.map(k => String((k as Keyframe & { transform?: string }).transform ?? ''))
|
||||
.join(' | ')
|
||||
}
|
||||
}, STRIP)
|
||||
|
||||
// The strip carries every frame; `steps(N)` parks on each one in turn.
|
||||
expect(observed.frameCount).toBeGreaterThan(1)
|
||||
// Chromium has serialized jump-end as both `steps(N)` and `steps(N, end)`.
|
||||
expect(observed.timingFunction).toMatch(new RegExp(`^steps\\(${observed.frameCount}\\b`))
|
||||
expect(observed.iterationCount).toBe('infinite')
|
||||
expect(observed.names).toContain('glyph-spinner-advance')
|
||||
// One full cycle is frames x interval, so the duration must be a positive
|
||||
// multiple of the frame count — not the single-frame interval.
|
||||
expect(observed.durationMs).toBeGreaterThan(0)
|
||||
// Length-typed travel, never a percentage: `translateY(-100%)` would keep
|
||||
// the animation off the compositor.
|
||||
expect(observed.travel).toContain('calc(')
|
||||
expect(observed.travel).not.toContain('%')
|
||||
})
|
||||
|
||||
test('is promoted to a layer while running, and neither animates nor holds a layer when parked', async () => {
|
||||
const { page } = fixture
|
||||
await mountSpinner(page)
|
||||
|
||||
const running = await page.evaluate(strip => {
|
||||
const el = document.querySelector<HTMLElement>(strip)!
|
||||
|
||||
return {
|
||||
playState: getComputedStyle(el).animationPlayState,
|
||||
willChange: getComputedStyle(el).willChange
|
||||
}
|
||||
}, STRIP)
|
||||
|
||||
expect(running.playState).toBe('running')
|
||||
// Scoped to active spinners — a permanently promoted layer per parked
|
||||
// spinner is pure memory at fan-out breadth.
|
||||
expect(running.willChange).toBe('transform')
|
||||
|
||||
// 1. The per-spinner gate: a kept-alive but inactive pane, or an explicit
|
||||
// `paused` prop (ChatSwapOverlay's fade-out).
|
||||
const parked = await page.evaluate(strip => {
|
||||
const el = document.querySelector<HTMLElement>(strip)!
|
||||
const viewport = el.closest<HTMLElement>('.glyph-spinner')!
|
||||
const previous = viewport.getAttribute('data-paused')
|
||||
|
||||
viewport.setAttribute('data-paused', 'true')
|
||||
|
||||
const state = {
|
||||
playState: getComputedStyle(el).animationPlayState,
|
||||
willChange: getComputedStyle(el).willChange
|
||||
}
|
||||
|
||||
if (previous === null) {
|
||||
viewport.removeAttribute('data-paused')
|
||||
} else {
|
||||
viewport.setAttribute('data-paused', previous)
|
||||
}
|
||||
|
||||
return state
|
||||
}, STRIP)
|
||||
|
||||
expect(parked.playState).toBe('paused')
|
||||
expect(parked.willChange).toBe('auto')
|
||||
|
||||
// 2. The global gate: window blur / minimize / document-hidden, which
|
||||
// main.tsx drives by arming this attribute on the root. The strip must
|
||||
// be named in that rule, or every spinner keeps animating behind an
|
||||
// inactive window — the CPU burn the original ticker's pause
|
||||
// controller existed to avoid.
|
||||
const globallyPaused = await page.evaluate(strip => {
|
||||
const root = document.documentElement
|
||||
const had = root.hasAttribute('data-renderer-animations-paused')
|
||||
|
||||
root.setAttribute('data-renderer-animations-paused', '')
|
||||
const playState = getComputedStyle(document.querySelector<HTMLElement>(strip)!).animationPlayState
|
||||
|
||||
if (!had) {
|
||||
root.removeAttribute('data-renderer-animations-paused')
|
||||
}
|
||||
|
||||
return playState
|
||||
}, STRIP)
|
||||
|
||||
expect(globallyPaused).toBe('paused')
|
||||
})
|
||||
|
||||
test('advances in discrete frames and creates no timer-driven DOM churn', async () => {
|
||||
const { page } = fixture
|
||||
await mountSpinner(page)
|
||||
|
||||
// Sample the resolved transform across one full cycle. A steps() animation
|
||||
// holds each value for a whole interval and jumps between them, so the
|
||||
// distinct values it visits must be bounded by the frame count — a linear
|
||||
// animation would produce a new value on every sample.
|
||||
const sampled = await page.evaluate(async strip => {
|
||||
const el = document.querySelector<HTMLElement>(strip)!
|
||||
const frames = el.querySelectorAll('.glyph-spinner__frame').length
|
||||
const duration = Number(el.getAnimations()[0]?.effect?.getTiming().duration ?? 0)
|
||||
const seen = new Set<string>()
|
||||
const textAtStart = el.textContent
|
||||
|
||||
const deadline = performance.now() + duration
|
||||
|
||||
while (performance.now() < deadline) {
|
||||
seen.add(getComputedStyle(el).transform)
|
||||
await new Promise(resolve => requestAnimationFrame(() => resolve(null)))
|
||||
}
|
||||
|
||||
return { distinct: seen.size, frames, textUnchanged: el.textContent === textAtStart }
|
||||
}, STRIP)
|
||||
|
||||
expect(sampled.distinct).toBeGreaterThan(1)
|
||||
expect(sampled.distinct).toBeLessThanOrEqual(sampled.frames + 1)
|
||||
// The old implementation rewrote textContent ~12x/second. Nothing may
|
||||
// mutate the DOM as this animates — that mutation is the whole incident.
|
||||
expect(sampled.textUnchanged).toBe(true)
|
||||
})
|
||||
})
|
||||
@@ -353,7 +353,7 @@ import { installWindowsSystemCaTrust } from './windows-system-ca'
|
||||
import { readWindowsUserEnvVar } from './windows-user-env'
|
||||
import { isPackagedInstallPath as isPackagedInstallPathUnderRoots } from './workspace-cwd'
|
||||
import { readWslWindowsClipboardImage } from './wsl-clipboard-image'
|
||||
import { resolvePickerDefaultPath } from './wsl-path-bridge'
|
||||
import { resolvePickerDefaultPath, setActiveGatewayProfile, setWslBridgeProfileState } from './wsl-path-bridge'
|
||||
|
||||
const USER_DATA_OVERRIDE = process.env.HERMES_DESKTOP_USER_DATA_DIR
|
||||
|
||||
@@ -9898,6 +9898,7 @@ async function ensureBackend(profile) {
|
||||
|
||||
if (route.backend === 'primary') {
|
||||
const connection = await startHermes()
|
||||
setWslBridgeProfileState(key, connection.mode !== 'remote')
|
||||
|
||||
// A shared backend still owes the caller its profile scope, so renderer-side
|
||||
// WebSocket, filesystem, and cache routing target the selected profile.
|
||||
@@ -9921,8 +9922,10 @@ async function ensureBackend(profile) {
|
||||
|
||||
if (existing) {
|
||||
existing.lastActiveAt = Date.now()
|
||||
const connection = await existing.connectionPromise
|
||||
setWslBridgeProfileState(key, connection.mode !== 'remote')
|
||||
|
||||
return existing.connectionPromise
|
||||
return connection
|
||||
}
|
||||
|
||||
evictLruPoolBackends(POOL_MAX_BACKENDS - 1)
|
||||
@@ -9955,7 +9958,10 @@ async function ensureBackend(profile) {
|
||||
backendPool.set(key, entry)
|
||||
startPoolIdleReaper()
|
||||
|
||||
return entry.connectionPromise
|
||||
const connection = await entry.connectionPromise
|
||||
setWslBridgeProfileState(key, connection.mode !== 'remote')
|
||||
|
||||
return connection
|
||||
}
|
||||
|
||||
// ── Registry-scoped backends (multi-connection, PR 2 of the campaign) ──────
|
||||
@@ -10579,6 +10585,11 @@ async function startHermes() {
|
||||
}
|
||||
|
||||
const connectionAttempt = backendConnectionState.startAttempt()
|
||||
const primaryProfile = primaryProfileKey()
|
||||
|
||||
// Legacy path callers without an explicit profile belong to the primary
|
||||
// window backend. Profile-scoped callers still pass their key directly.
|
||||
setActiveGatewayProfile(primaryProfile)
|
||||
|
||||
// Classify this boot BEFORE the throwing resolve/mint runs: a remote failure
|
||||
// must NOT latch (it's transient — see shouldLatchBackendStartFailure), while
|
||||
@@ -10660,7 +10671,7 @@ async function startHermes() {
|
||||
// both for an already-saved remote and after first-run remote Apply.
|
||||
attemptedRemote = primaryBackendIsRemote()
|
||||
|
||||
return resolveRemoteBackend(primaryProfileKey())
|
||||
return resolveRemoteBackend(primaryProfile)
|
||||
},
|
||||
waitForDecision: waitForFirstRunSetupChoice,
|
||||
// Mutual exclusion with an in-app update (#50238). Remote connections
|
||||
@@ -10669,9 +10680,18 @@ async function startHermes() {
|
||||
})
|
||||
|
||||
if (setup.kind === 'remote') {
|
||||
// Paths from the remote backend belong to a host the Windows desktop
|
||||
// cannot open via wsl.exe — disable WSL path bridging so native dialogs
|
||||
// and file panels don't spawn wsl.exe (or the interactive install prompt
|
||||
// on WSL-less machines) for unresolvable paths. (#66433)
|
||||
setWslBridgeProfileState(primaryProfile, false)
|
||||
|
||||
return setup.connection
|
||||
}
|
||||
|
||||
// Local WSL backend — paths are bridgeable.
|
||||
setWslBridgeProfileState(primaryProfile, true)
|
||||
|
||||
const backend = setup.backend
|
||||
// Route old runtimes (no `serve`) through the legacy `dashboard --no-open`.
|
||||
backend.args = getBackendArgsForRuntime(backend)
|
||||
@@ -13940,7 +13960,10 @@ ipcMain.handle('hermes:selectPaths', async (_event, options: any = {}) => {
|
||||
try {
|
||||
// On a Windows host with a WSL backend the cwd may be a POSIX/WSL path;
|
||||
// bridge it to a UNC/drive form the native dialog can actually open.
|
||||
const bridged = IS_WINDOWS ? resolvePickerDefaultPath(String(options.defaultPath)) : String(options.defaultPath)
|
||||
const bridged = IS_WINDOWS
|
||||
? resolvePickerDefaultPath(String(options.defaultPath), undefined, options?.profile)
|
||||
: String(options.defaultPath)
|
||||
|
||||
resolvedDefaultPath = bridged ? path.resolve(bridged) : undefined
|
||||
} catch {
|
||||
resolvedDefaultPath = undefined
|
||||
@@ -15030,6 +15053,12 @@ app.whenReady().then(() => {
|
||||
registerPowerResumeListeners()
|
||||
keepAwake.set(readPersistedKeepAwake())
|
||||
f12Blocked = readPersistedDisableF12()
|
||||
// Seed this before the first window exists: a picker can open before
|
||||
// startHermes() finishes resolving the configured backend.
|
||||
const primaryProfile = primaryProfileKey()
|
||||
|
||||
setActiveGatewayProfile(primaryProfile)
|
||||
setWslBridgeProfileState(primaryProfile, !primaryBackendIsRemote())
|
||||
// Quick Entry's global chord — registered on ready so a cold launch restores
|
||||
// it without the renderer visiting Settings. A failed registration is logged
|
||||
// here and surfaced in Settings via the IPC state (never silent).
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
/**
|
||||
* Windows-platform regression for the WSL path-bridge gate (#66433).
|
||||
*
|
||||
* The behavioural tests in wsl-path-bridge.test.ts prove the no-op contract
|
||||
* (paths pass through unchanged when the bridge is inactive). This file goes
|
||||
* one rung further: with `process.platform` stubbed to `win32` and
|
||||
* `child_process.execFileSync` mocked, it proves the actual `wsl.exe` spawn is
|
||||
* suppressed — not just that the return value looks right.
|
||||
*
|
||||
* Each test re-imports the module fresh (vi.resetModules) so IS_WINDOWS is
|
||||
* re-evaluated against the stubbed platform.
|
||||
*/
|
||||
import { afterEach, beforeEach, describe, expect, test, vi } from 'vitest'
|
||||
|
||||
const execFileSyncMock = vi.fn(() => 'Ubuntu\n')
|
||||
|
||||
vi.mock('node:child_process', () => ({ execFileSync: execFileSyncMock }))
|
||||
|
||||
describe('WSL bridge gate on Windows (#66433)', () => {
|
||||
const realPlatform = process.platform
|
||||
|
||||
beforeEach(() => {
|
||||
Object.defineProperty(process, 'platform', { value: 'win32', configurable: true })
|
||||
vi.resetModules()
|
||||
execFileSyncMock.mockClear()
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
Object.defineProperty(process, 'platform', { value: realPlatform, configurable: true })
|
||||
})
|
||||
|
||||
test('wsl.exe IS probed for a POSIX path when the bridge is active (control)', async () => {
|
||||
const { resolveLocalReadPath } = await import('./wsl-path-bridge')
|
||||
resolveLocalReadPath('/home/ubuntu/project')
|
||||
expect(execFileSyncMock).toHaveBeenCalled()
|
||||
// Sanity: it really was wsl.exe, not some other binary.
|
||||
expect(execFileSyncMock).toHaveBeenNthCalledWith(
|
||||
1,
|
||||
'wsl.exe',
|
||||
expect.arrayContaining(['-l', '-q']),
|
||||
expect.anything()
|
||||
)
|
||||
})
|
||||
|
||||
test('wsl.exe is NEVER probed when the bridge is inactive — even for POSIX paths', async () => {
|
||||
const { resolveLocalReadPath, setWslBridgeActive } = await import('./wsl-path-bridge')
|
||||
setWslBridgeActive(false)
|
||||
// A POSIX path that WOULD trigger bridging (and the wsl.exe probe) when
|
||||
// active — but with the bridge off, resolveDefaultWslDistro is never
|
||||
// reached because resolveLocalReadPath returns before it.
|
||||
const result = resolveLocalReadPath('/home/ubuntu/project')
|
||||
expect(execFileSyncMock).not.toHaveBeenCalled()
|
||||
expect(result).toBe('/home/ubuntu/project')
|
||||
})
|
||||
|
||||
test('the picker default-path also skips the wsl.exe probe when inactive', async () => {
|
||||
const { resolvePickerDefaultPath, setWslBridgeActive } = await import('./wsl-path-bridge')
|
||||
setWslBridgeActive(false)
|
||||
const result = resolvePickerDefaultPath('/home/ubuntu')
|
||||
expect(execFileSyncMock).not.toHaveBeenCalled()
|
||||
expect(result).toBe('/home/ubuntu')
|
||||
})
|
||||
|
||||
test('re-enabling the bridge restores wsl.exe probing', async () => {
|
||||
const { resolveLocalReadPath, setWslBridgeActive } = await import('./wsl-path-bridge')
|
||||
setWslBridgeActive(false)
|
||||
resolveLocalReadPath('/home/ubuntu/project')
|
||||
expect(execFileSyncMock).not.toHaveBeenCalled()
|
||||
|
||||
setWslBridgeActive(true)
|
||||
execFileSyncMock.mockClear()
|
||||
resolveLocalReadPath('/home/ubuntu/project')
|
||||
expect(execFileSyncMock).toHaveBeenCalled()
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,242 @@
|
||||
/**
|
||||
* Profile-scoped eligibility for the WSL path bridge (#66447).
|
||||
*
|
||||
* The single-profile tests in wsl-path-bridge.test.ts and the Windows-platform
|
||||
* gate tests in wsl-path-bridge-gate.test.ts cover the *what* (paths pass
|
||||
* through unchanged when bridging is disabled) but not the *which profile*. The
|
||||
* desktop is multi-profile: the renderer can swap the live gateway onto any
|
||||
* profile (primary or pool) without reloading the window — so the bridge
|
||||
* eligibility MUST be keyed off the **currently active profile's** backend
|
||||
* configuration, not a process-global boolean. This file proves the
|
||||
* per-profile contract:
|
||||
*
|
||||
* 1. local primary profile → bridge ON (preserved)
|
||||
* 2. remote primary profile → bridge OFF (preserved)
|
||||
* 3. local primary + remote non-primary → bridge OFF when the non-primary
|
||||
* is active; bridge ON when the primary is active again — **no bleed**.
|
||||
* 4. remote primary + local non-primary → bridge ON when the non-primary
|
||||
* is active; bridge OFF when the primary is active again — **no bleed**.
|
||||
* 5. profile-scoped calls accept a profile argument; absent it falls back
|
||||
* to the live gateway profile that the renderer announced.
|
||||
* 6. afterEach resets state so tests don't bleed into each other.
|
||||
*
|
||||
* These tests are pure behavior: they assert the eligibility function's output
|
||||
* and the public selectors' output against observable calls. They do NOT read
|
||||
* the implementation source — only the public surface exported from
|
||||
* `./wsl-path-bridge`.
|
||||
*/
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import { afterEach, describe, test } from 'vitest'
|
||||
|
||||
import {
|
||||
isWslBridgeActive,
|
||||
resolveLocalReadPath,
|
||||
resolvePickerDefaultPath,
|
||||
setActiveGatewayProfile,
|
||||
setWslBridgeActive,
|
||||
setWslBridgeProfileState,
|
||||
wslPosixToWindowsAccessible
|
||||
} from './wsl-path-bridge'
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────
|
||||
|
||||
const PROFILE_PRIMARY = 'default'
|
||||
const PROFILE_LOCAL = 'team-local'
|
||||
const PROFILE_REMOTE = 'team-remote'
|
||||
|
||||
/** Reset every profile key the bridge knows about to a clean state. */
|
||||
afterEach(() => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, true)
|
||||
setWslBridgeProfileState(PROFILE_LOCAL, true)
|
||||
setWslBridgeProfileState(PROFILE_REMOTE, false)
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
setWslBridgeActive(true)
|
||||
})
|
||||
|
||||
// ── single-profile contract (preserved behaviour) ────────────────────
|
||||
|
||||
describe('WSL bridge profile eligibility — single-profile contract preserved', () => {
|
||||
test('primary local → bridge ON (preserved)', () => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, true)
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
assert.equal(isWslBridgeActive(), true)
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_PRIMARY),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
})
|
||||
|
||||
test('primary remote → bridge OFF (preserved)', () => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, false)
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
assert.equal(isWslBridgeActive(), false)
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_PRIMARY), '/home/alex')
|
||||
assert.equal(resolveLocalReadPath('/home/alex/proj', undefined, PROFILE_PRIMARY), '/home/alex/proj')
|
||||
})
|
||||
})
|
||||
|
||||
// ── multi-profile regression (the gap) ────────────────────────────────
|
||||
|
||||
describe('WSL bridge profile eligibility — multi-profile (no cross-profile bleed)', () => {
|
||||
test('local primary + remote non-primary → non-primary OFF, primary ON', () => {
|
||||
// Primary is a local backend (WSL on this Windows host). A second profile
|
||||
// points at a remote host whose POSIX paths the Windows host CANNOT open
|
||||
// via wsl.exe — bridging must be OFF for it.
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, true)
|
||||
setWslBridgeProfileState(PROFILE_REMOTE, false)
|
||||
|
||||
// Non-primary remote is foregrounded — bridge must be OFF for it.
|
||||
setActiveGatewayProfile(PROFILE_REMOTE)
|
||||
assert.equal(isWslBridgeActive(), false)
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_REMOTE), '/home/alex')
|
||||
assert.equal(resolveLocalReadPath('/home/alex/proj', undefined, PROFILE_REMOTE), '/home/alex/proj')
|
||||
|
||||
// Swap back to local primary — bridge must be ON again, no bleed.
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
assert.equal(isWslBridgeActive(), true)
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_PRIMARY),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
})
|
||||
|
||||
test('remote primary + local non-primary → non-primary ON, primary OFF', () => {
|
||||
// Primary is remote (no local WSL paths). A second profile is local —
|
||||
// bridging should be ON for it because its paths CAN be opened locally.
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, false)
|
||||
setWslBridgeProfileState(PROFILE_LOCAL, true)
|
||||
|
||||
// Local non-primary foregrounded — bridge ON for it.
|
||||
setActiveGatewayProfile(PROFILE_LOCAL)
|
||||
assert.equal(isWslBridgeActive(), true)
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_LOCAL),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
|
||||
// Swap back to remote primary — bridge OFF for it, no bleed.
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
assert.equal(isWslBridgeActive(), false)
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_PRIMARY), '/home/alex')
|
||||
})
|
||||
|
||||
test('three profiles: each profile behaves independently', () => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, true)
|
||||
setWslBridgeProfileState(PROFILE_LOCAL, true)
|
||||
setWslBridgeProfileState(PROFILE_REMOTE, false)
|
||||
|
||||
// Same path, different profiles, different outcomes.
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_PRIMARY),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_LOCAL),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_REMOTE), '/home/alex')
|
||||
})
|
||||
})
|
||||
|
||||
// ── profile-argument contract ────────────────────────────────────────
|
||||
|
||||
describe('WSL bridge profile eligibility — selector argument contract', () => {
|
||||
test("selector with explicit profile → that profile's bridge state", () => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, true)
|
||||
setWslBridgeProfileState(PROFILE_REMOTE, false)
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
|
||||
// Explicit profile wins over the live gateway.
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_REMOTE), '/home/alex')
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_PRIMARY),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
})
|
||||
|
||||
test("selector with no profile → live gateway profile's bridge state", () => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, true)
|
||||
setWslBridgeProfileState(PROFILE_REMOTE, false)
|
||||
setActiveGatewayProfile(PROFILE_REMOTE)
|
||||
|
||||
// No profile argument → fallback to active gateway profile (remote → OFF).
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu'), '/home/alex')
|
||||
assert.equal(resolveLocalReadPath('/home/alex/proj'), '/home/alex/proj')
|
||||
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu'), '\\\\wsl.localhost\\Ubuntu\\home\\alex')
|
||||
})
|
||||
|
||||
test('selector with unknown profile → bridge ON (defaults to active for new profiles)', () => {
|
||||
// A profile that has never been seeded should default to the safe
|
||||
// "bridge ON" behaviour so a brand-new local profile isn't accidentally
|
||||
// disabled. The renderer seeds the state at boot; an unknown key here is
|
||||
// either a renderer race or a profile created mid-session.
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, false)
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', 'unknown-profile'),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
// ── legacy back-compat: setWslBridgeActive targets the active profile ─
|
||||
|
||||
describe('WSL bridge profile eligibility — legacy toggle targets active profile', () => {
|
||||
test('setWslBridgeActive(false) flips the active profile, not a process-global', () => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, true)
|
||||
setWslBridgeProfileState(PROFILE_REMOTE, true)
|
||||
setActiveGatewayProfile(PROFILE_REMOTE)
|
||||
|
||||
setWslBridgeActive(false)
|
||||
|
||||
// Active profile (REMOTE) flipped to OFF; primary untouched.
|
||||
assert.equal(isWslBridgeActive(), false)
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_PRIMARY),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_REMOTE), '/home/alex')
|
||||
})
|
||||
|
||||
test('setWslBridgeActive(true) restores the active profile only', () => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, true)
|
||||
setWslBridgeProfileState(PROFILE_REMOTE, false)
|
||||
setActiveGatewayProfile(PROFILE_REMOTE)
|
||||
|
||||
setWslBridgeActive(true)
|
||||
|
||||
// Active (REMOTE) restored; primary unchanged.
|
||||
assert.equal(isWslBridgeActive(), true)
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_REMOTE),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
assert.equal(
|
||||
resolvePickerDefaultPath('/home/alex', 'Ubuntu', PROFILE_PRIMARY),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex'
|
||||
)
|
||||
})
|
||||
})
|
||||
|
||||
// ── wslPosixToWindowsAccessible stays pure / unchanged ───────────────
|
||||
|
||||
describe('WSL bridge profile eligibility — POSIX translation stays pure', () => {
|
||||
test('wslPosixToWindowsAccessible ignores bridge state (pure translation)', () => {
|
||||
setWslBridgeProfileState(PROFILE_PRIMARY, false)
|
||||
setActiveGatewayProfile(PROFILE_PRIMARY)
|
||||
|
||||
// The translator does NOT consult the bridge state — it's pure POSIX → UNC.
|
||||
// Tests elsewhere assert that the bridge gate short-circuits BEFORE this
|
||||
// function is reached. A regression here would mean coupling leaked into
|
||||
// a helper that should stay side-effect-free.
|
||||
assert.equal(
|
||||
wslPosixToWindowsAccessible('/home/alex/proj', 'Ubuntu'),
|
||||
'\\\\wsl.localhost\\Ubuntu\\home\\alex\\proj'
|
||||
)
|
||||
assert.equal(wslPosixToWindowsAccessible('/mnt/c/Users/alex', 'Ubuntu'), 'C:\\Users\\alex')
|
||||
})
|
||||
})
|
||||
@@ -1,8 +1,25 @@
|
||||
import assert from 'node:assert/strict'
|
||||
|
||||
import { test } from 'vitest'
|
||||
import { afterEach, test } from 'vitest'
|
||||
|
||||
import { parseDefaultDistro, resolvePickerDefaultPath, wslPosixToWindowsAccessible } from './wsl-path-bridge'
|
||||
import {
|
||||
isWslBridgeActive,
|
||||
parseDefaultDistro,
|
||||
resolveLocalReadPath,
|
||||
resolvePickerDefaultPath,
|
||||
setWslBridgeActive,
|
||||
wslPosixToWindowsAccessible
|
||||
} from './wsl-path-bridge'
|
||||
|
||||
// ── helpers ──────────────────────────────────────────────────────────
|
||||
|
||||
/** Reset the bridge to its default active state after every test so no test
|
||||
* leaks global state into the next one. */
|
||||
afterEach(() => {
|
||||
setWslBridgeActive(true)
|
||||
})
|
||||
|
||||
// ── distro parsing (unchanged) ───────────────────────────────────────
|
||||
|
||||
test('parseDefaultDistro reads the first distro from clean utf-8 output', () => {
|
||||
assert.equal(parseDefaultDistro('Ubuntu\nDebian\n'), 'Ubuntu')
|
||||
@@ -20,6 +37,8 @@ test('parseDefaultDistro strips the default-marker and blank lines', () => {
|
||||
assert.equal(parseDefaultDistro(' \n\n'), null)
|
||||
})
|
||||
|
||||
// ── wslPosixToWindowsAccessible ──────────────────────────────────────
|
||||
|
||||
test('wslPosixToWindowsAccessible maps a drvfs mount to its Windows drive', () => {
|
||||
assert.equal(wslPosixToWindowsAccessible('/mnt/c/Users/alex', 'Ubuntu'), 'C:\\Users\\alex')
|
||||
assert.equal(wslPosixToWindowsAccessible('/mnt/d', 'Ubuntu'), 'D:\\')
|
||||
@@ -34,8 +53,73 @@ test('wslPosixToWindowsAccessible leaves non-absolute / already-Windows paths al
|
||||
assert.equal(wslPosixToWindowsAccessible('relative/dir', 'Ubuntu'), 'relative/dir')
|
||||
})
|
||||
|
||||
// ── resolvePickerDefaultPath (bridge active) ─────────────────────────
|
||||
|
||||
test('resolvePickerDefaultPath bridges a WSL cwd but passes Windows paths and empties through', () => {
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu'), '\\\\wsl.localhost\\Ubuntu\\home\\alex')
|
||||
assert.equal(resolvePickerDefaultPath('C:\\proj', 'Ubuntu'), 'C:\\proj')
|
||||
assert.equal(resolvePickerDefaultPath(undefined, 'Ubuntu'), undefined)
|
||||
})
|
||||
|
||||
// ── bridge active / inactive ─────────────────────────────────────────
|
||||
|
||||
test('bridge defaults to active', () => {
|
||||
assert.equal(isWslBridgeActive(), true)
|
||||
})
|
||||
|
||||
test('setWslBridgeActive(false) → resolvePickerDefaultPath passes raw path through without bridging', () => {
|
||||
setWslBridgeActive(false)
|
||||
// Even a clear WSL POSIX path must pass through unchanged when the bridge
|
||||
// is inactive — no distro probe, no wsl.exe, no install prompt.
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex'), '/home/alex')
|
||||
assert.equal(resolvePickerDefaultPath('/mnt/c/Users/alex'), '/mnt/c/Users/alex')
|
||||
// Windows paths and empties are unaffected either way.
|
||||
assert.equal(resolvePickerDefaultPath('C:\\proj'), 'C:\\proj')
|
||||
assert.equal(resolvePickerDefaultPath(undefined), undefined)
|
||||
})
|
||||
|
||||
test('setWslBridgeActive(false) → resolveLocalReadPath passes raw path through without bridging', () => {
|
||||
setWslBridgeActive(false)
|
||||
// resolveLocalReadPath is used by fs-read-dir to make WSL paths readable
|
||||
// on the Windows host. When the bridge is inactive (remote gateway), the
|
||||
// raw POSIX path must be returned as-is — no UNC rewriting, no distro
|
||||
// resolution. The downstream fs call will fail gracefully on non-WSL
|
||||
// hosts, which is the desired behaviour.
|
||||
assert.equal(resolveLocalReadPath('/home/alex/proj'), '/home/alex/proj')
|
||||
assert.equal(resolveLocalReadPath('/mnt/c/Users/alex'), '/mnt/c/Users/alex')
|
||||
// Non-POSIX paths are never bridged regardless of state.
|
||||
assert.equal(resolveLocalReadPath('C:\\Users\\alex'), 'C:\\Users\\alex')
|
||||
assert.equal(resolveLocalReadPath(''), '')
|
||||
})
|
||||
|
||||
test('setWslBridgeActive(true) restores picker bridging', () => {
|
||||
setWslBridgeActive(false)
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex'), '/home/alex')
|
||||
|
||||
setWslBridgeActive(true)
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu'), '\\\\wsl.localhost\\Ubuntu\\home\\alex')
|
||||
})
|
||||
|
||||
test('toggling the bridge is idempotent and does not corrupt cached state', () => {
|
||||
// Toggle twice each way.
|
||||
setWslBridgeActive(false)
|
||||
assert.equal(isWslBridgeActive(), false)
|
||||
setWslBridgeActive(false)
|
||||
assert.equal(isWslBridgeActive(), false)
|
||||
|
||||
setWslBridgeActive(true)
|
||||
assert.equal(isWslBridgeActive(), true)
|
||||
setWslBridgeActive(true)
|
||||
assert.equal(isWslBridgeActive(), true)
|
||||
|
||||
// Bridging still works after the toggles.
|
||||
assert.equal(resolvePickerDefaultPath('/home/alex', 'Ubuntu'), '\\\\wsl.localhost\\Ubuntu\\home\\alex')
|
||||
})
|
||||
|
||||
// ── state isolation: every test sees a clean active bridge ────────────
|
||||
|
||||
test('state isolation: bridge is active after a previous test toggled it off', () => {
|
||||
// This test relies on afterEach resetting the bridge.
|
||||
// If isolation is broken, isWslBridgeActive() would be false here.
|
||||
assert.equal(isWslBridgeActive(), true)
|
||||
})
|
||||
|
||||
@@ -16,6 +16,42 @@ const WSL_MOUNT_RE = /^\/mnt\/([a-z])(?:\/(.*))?$/i
|
||||
let cachedDistro: null | string = null
|
||||
let cachedUncBase: null | string = null
|
||||
|
||||
/**
|
||||
* WSL path eligibility belongs to the backend profile that produced the path.
|
||||
* A single desktop process can keep a local primary backend and a remote pool
|
||||
* backend alive simultaneously, so a process-global boolean can bleed between
|
||||
* them. Unknown profiles retain the historical local default until Electron
|
||||
* resolves and records their actual backend mode.
|
||||
*/
|
||||
const DEFAULT_WSL_BRIDGE_PROFILE = 'default'
|
||||
const wslBridgeProfiles = new Map<string, boolean>()
|
||||
let activeWslBridgeProfile = DEFAULT_WSL_BRIDGE_PROFILE
|
||||
|
||||
function normalizeWslBridgeProfile(profile?: null | string): string {
|
||||
return String(profile || '').trim() || DEFAULT_WSL_BRIDGE_PROFILE
|
||||
}
|
||||
|
||||
/** Select the profile used by legacy callers that cannot pass one explicitly. */
|
||||
export function setActiveGatewayProfile(profile?: null | string): void {
|
||||
activeWslBridgeProfile = normalizeWslBridgeProfile(profile)
|
||||
}
|
||||
|
||||
/** Record whether paths returned by one profile belong to this host's WSL. */
|
||||
export function setWslBridgeProfileState(profile: null | string, active: boolean): void {
|
||||
wslBridgeProfiles.set(normalizeWslBridgeProfile(profile), Boolean(active))
|
||||
}
|
||||
|
||||
/** Backward-compatible toggle: update only the current fallback profile. */
|
||||
export function setWslBridgeActive(active: boolean): void {
|
||||
setWslBridgeProfileState(activeWslBridgeProfile, active)
|
||||
}
|
||||
|
||||
export function isWslBridgeActive(profile?: null | string): boolean {
|
||||
const key = profile == null ? activeWslBridgeProfile : normalizeWslBridgeProfile(profile)
|
||||
|
||||
return wslBridgeProfiles.get(key) ?? true
|
||||
}
|
||||
|
||||
/**
|
||||
* Pick the default distro from `wsl.exe -l -q` output.
|
||||
*
|
||||
@@ -50,6 +86,11 @@ export function resolveDefaultWslDistro(): string {
|
||||
const out = execFileSync('wsl.exe', ['-l', '-q'], {
|
||||
encoding: 'utf8',
|
||||
env: { ...process.env, WSL_UTF8: '1' },
|
||||
// On WSL-less machines wsl.exe prints "The Windows Subsystem for Linux
|
||||
// is not installed..." to stderr; stderr is inherited by default, so
|
||||
// that banner leaks into whatever console the app is attached to
|
||||
// (visible e.g. during the update hand-off). Discard it. (#80184)
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
timeout: 2000,
|
||||
windowsHide: true
|
||||
})
|
||||
@@ -116,22 +157,40 @@ export function wslPosixToWindowsAccessible(posixPath: string, distro: string =
|
||||
/** Native folder dialog `defaultPath`: open a WSL cwd in the Windows picker. */
|
||||
export function resolvePickerDefaultPath(
|
||||
defaultPath: string | undefined,
|
||||
distro: string = resolveDefaultWslDistro()
|
||||
distro?: string,
|
||||
profile?: null | string
|
||||
): string | undefined {
|
||||
if (!defaultPath) {
|
||||
return undefined
|
||||
}
|
||||
|
||||
// Remote-gateway POSIX paths can't be opened via wsl.exe — no-op the bridge
|
||||
// so the native dialog gets the raw path (it falls back gracefully) instead
|
||||
// of triggering a wsl.exe spawn / install prompt. (#66433)
|
||||
if (!isWslBridgeActive(profile)) {
|
||||
return defaultPath
|
||||
}
|
||||
|
||||
const value = String(defaultPath).trim()
|
||||
|
||||
return value.startsWith('/') && !WIN_DRIVE_RE.test(value) ? wslPosixToWindowsAccessible(value, distro) : defaultPath
|
||||
return value.startsWith('/') && !WIN_DRIVE_RE.test(value)
|
||||
? wslPosixToWindowsAccessible(value, distro ?? resolveDefaultWslDistro())
|
||||
: defaultPath
|
||||
}
|
||||
|
||||
/** fs read path: on Windows, make a WSL cwd readable via its UNC / drive form. */
|
||||
export function resolveLocalReadPath(dirPath: string, distro: string = resolveDefaultWslDistro()): string {
|
||||
export function resolveLocalReadPath(dirPath: string, distro?: string, profile?: null | string): string {
|
||||
const value = String(dirPath || '').trim()
|
||||
|
||||
// In remote-gateway mode the POSIX paths belong to a host the Windows
|
||||
// desktop cannot open locally — skip the WSL bridge entirely (no distro
|
||||
// probe, no wsl.exe) so the file panel never spawns the install prompt on
|
||||
// WSL-less machines. (#66433)
|
||||
if (!isWslBridgeActive(profile)) {
|
||||
return value
|
||||
}
|
||||
|
||||
return IS_WINDOWS && value.startsWith('/') && !WIN_DRIVE_RE.test(value)
|
||||
? wslPosixToWindowsAccessible(value, distro)
|
||||
? wslPosixToWindowsAccessible(value, distro ?? resolveDefaultWslDistro())
|
||||
: value
|
||||
}
|
||||
|
||||
@@ -925,7 +925,19 @@ async function runRealChatChecks(cdp, timed, measure, mock, runDir, appPid, nati
|
||||
const streamingRequestsBefore = mock.streamingCompletionRequests()
|
||||
const beforeAssistant = await timed(`real-chat.assistant-count.${exchange}`, () =>
|
||||
cdp.eval(
|
||||
`document.querySelectorAll('[data-slot="aui_assistant-message-root"]:not([data-streaming="true"])').length`
|
||||
// Settled = no streaming marker anywhere in the row's subtree. The
|
||||
// marker moved off the message root onto a hidden leaf inside it — a
|
||||
// per-flip attribute write on the root widened style recalc across the
|
||||
// whole message subtree — so this counts the descendant instead of the
|
||||
// root's own attribute. Same rows, same gate.
|
||||
//
|
||||
// Counted by subtraction rather than with
|
||||
// `:not(:has([data-message-streaming="true"]))`, which makes the engine
|
||||
// walk every row's subtree on each evaluation — and this probe runs
|
||||
// inside the very latency window it is measuring, so that cost lands in
|
||||
// the number. At most one marker per row carries the attribute, so
|
||||
// roots - streaming is exactly the settled count.
|
||||
`document.querySelectorAll('[data-slot="aui_assistant-message-root"]').length - document.querySelectorAll('[data-message-streaming="true"]').length`
|
||||
)
|
||||
)
|
||||
const composer = await timed(`real-chat.composer-focus.${exchange}`, () =>
|
||||
@@ -1020,7 +1032,9 @@ async function runRealChatChecks(cdp, timed, measure, mock, runDir, appPid, nati
|
||||
await measure(`real-chat.assistant-response.${exchange}`, () =>
|
||||
waitForResponsive(
|
||||
cdp,
|
||||
`document.querySelectorAll('[data-slot="aui_assistant-message-root"]:not([data-streaming="true"])').length > ${beforeAssistant}`,
|
||||
// Same count-by-subtraction as the pre-send probe above: no `:has()`
|
||||
// subtree walk inside the responsiveness measurement window.
|
||||
`document.querySelectorAll('[data-slot="aui_assistant-message-root"]').length - document.querySelectorAll('[data-message-streaming="true"]').length > ${beforeAssistant}`,
|
||||
STREAM_RESPONSE_TIMEOUT_MS,
|
||||
`real assistant response ${exchange}`,
|
||||
STREAM_RESPONSE_EVALUATION_TIMEOUT_MS
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
// The overlay used to run its own 80ms setInterval + setState braille ticker —
|
||||
// the same mechanism class (per-tick DOM mutation scheduling a style recalc)
|
||||
// that GlyphSpinner was rewritten to remove. It now renders GlyphSpinner, so
|
||||
// what needs pinning is that no timer comes back, that the label still survives
|
||||
// the fade-out, and that the spinner stops animating once the swap is done.
|
||||
import { cleanup, render, screen } from '@testing-library/react'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { ChatSwapOverlay } from './chat-swap-overlay'
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
})
|
||||
|
||||
describe('ChatSwapOverlay', () => {
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers()
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
vi.clearAllTimers()
|
||||
vi.useRealTimers()
|
||||
})
|
||||
|
||||
it('animates the glyph without any timer', () => {
|
||||
const { container } = render(<ChatSwapOverlay profile="turqoise" />)
|
||||
|
||||
expect(container.querySelector('.glyph-spinner__strip')).toBeTruthy()
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
vi.advanceTimersByTime(5_000)
|
||||
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
})
|
||||
|
||||
it('names the waking profile', () => {
|
||||
render(<ChatSwapOverlay profile="turqoise" />)
|
||||
|
||||
expect(screen.getByText(/turqoise/)).toBeTruthy()
|
||||
})
|
||||
|
||||
it('keeps the last profile name through the fade-out, with the glyph frozen', () => {
|
||||
const { container, rerender } = render(<ChatSwapOverlay profile="turqoise" />)
|
||||
|
||||
expect(container.querySelector('.glyph-spinner')?.hasAttribute('data-paused')).toBe(false)
|
||||
|
||||
rerender(<ChatSwapOverlay profile={null} />)
|
||||
|
||||
// Label held so the overlay doesn't blank while it fades.
|
||||
expect(screen.getByText(/turqoise/)).toBeTruthy()
|
||||
// ...and the spinner stops, the way clearing the interval used to stop it.
|
||||
expect(container.querySelector('.glyph-spinner')?.getAttribute('data-paused')).toBe('true')
|
||||
})
|
||||
})
|
||||
@@ -1,17 +1,14 @@
|
||||
import { useEffect, useState } from 'react'
|
||||
|
||||
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { cn } from '@/lib/utils'
|
||||
|
||||
// Braille spinner frames — reads as a tiny ASCII loader in monospace.
|
||||
const FRAMES = ['⠋', '⠙', '⠹', '⠸', '⠼', '⠴', '⠦', '⠧', '⠇', '⠏']
|
||||
|
||||
// Shown over the conversation while the live gateway swaps to another profile's
|
||||
// backend (lazily spawned). Keeps the last profile name through the fade-out so
|
||||
// the label doesn't blank. Purely visual — pointer-events-none.
|
||||
export function ChatSwapOverlay({ profile }: { profile: string | null }) {
|
||||
const { t } = useI18n()
|
||||
const [frame, setFrame] = useState(0)
|
||||
const [label, setLabel] = useState<null | string>(profile)
|
||||
|
||||
useEffect(() => {
|
||||
@@ -20,16 +17,6 @@ export function ChatSwapOverlay({ profile }: { profile: string | null }) {
|
||||
}
|
||||
}, [profile])
|
||||
|
||||
useEffect(() => {
|
||||
if (!profile) {
|
||||
return
|
||||
}
|
||||
|
||||
const id = window.setInterval(() => setFrame(value => (value + 1) % FRAMES.length), 80)
|
||||
|
||||
return () => window.clearInterval(id)
|
||||
}, [profile])
|
||||
|
||||
return (
|
||||
<div
|
||||
aria-hidden
|
||||
@@ -39,7 +26,14 @@ export function ChatSwapOverlay({ profile }: { profile: string | null }) {
|
||||
)}
|
||||
>
|
||||
<div className="flex items-center gap-2 bg-[color-mix(in_srgb,var(--dt-card)_92%,transparent)] px-4 py-2 font-mono text-[0.8125rem] text-foreground shadow-composer">
|
||||
<span className="w-3 text-(--ui-accent)">{FRAMES[frame]}</span>
|
||||
{/* Was a local 80ms setInterval + setState braille ticker — the same
|
||||
mechanism class (per-tick DOM mutation scheduling style recalc)
|
||||
that GlyphSpinner was rewritten to remove. `braille` is exactly the
|
||||
frame set and 80ms cadence this used. `justify-start` keeps the
|
||||
glyph left-aligned in its w-3 box the way the bare span was, and
|
||||
`paused` restores the old "no ticking once the swap is done"
|
||||
behaviour while the overlay fades out still mounted. */}
|
||||
<GlyphSpinner className="w-3 justify-start text-(--ui-accent)" paused={!profile} spinner="braille" />
|
||||
{t.composer.wakingProfile(label ?? '')}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -52,6 +52,7 @@ import {
|
||||
$attentionSessionIds,
|
||||
$workingSessionIds,
|
||||
liveSessionScopes,
|
||||
reconcileBusyStatesOnReconnect,
|
||||
recordSessionEventScope,
|
||||
resetTileRuntimeBindings
|
||||
} from '@/store/session-states'
|
||||
@@ -252,6 +253,12 @@ export function useGatewayBoot({
|
||||
// A respawned backend re-mints (recycles) runtime ids, so any tile's
|
||||
// bound runtime id is now stale — drop them so each tile re-resumes.
|
||||
resetTileRuntimeBindings()
|
||||
// Same staleness, other half: pre-reconnect busy flags are keyed by
|
||||
// those dead runtime ids and would never receive their terminal
|
||||
// busy:false — clear them or the sidebar running arc lies forever
|
||||
// (#53902/#73082). A genuinely live turn re-asserts busy on its next
|
||||
// post-reconnect event.
|
||||
reconcileBusyStatesOnReconnect()
|
||||
// Resync state that may have moved on the backend while we were asleep.
|
||||
await callbacksRef.current.refreshHermesConfig().catch(() => undefined)
|
||||
await callbacksRef.current.refreshSessions().catch(() => undefined)
|
||||
|
||||
@@ -7,7 +7,7 @@ import {
|
||||
useMessageRuntime
|
||||
} from '@assistant-ui/react'
|
||||
import { useStore } from '@nanostores/react'
|
||||
import { type FC, useCallback, useMemo, useState } from 'react'
|
||||
import { type FC, type ReactNode, useCallback, useMemo, useState } from 'react'
|
||||
|
||||
import { useSessionView } from '@/app/chat/session-view'
|
||||
import { ChangedFilesCard } from '@/components/assistant-ui/thread/changed-files-card'
|
||||
@@ -42,6 +42,14 @@ import { $voicePlayback } from '@/store/voice-playback'
|
||||
// would re-derive the changed-files card on every message re-render.
|
||||
const EMPTY_PARTS: readonly unknown[] = []
|
||||
|
||||
// PERF: hoisted to module scope so the element OBJECT is identical on every
|
||||
// render of every assistant message. React bails out of re-rendering a child
|
||||
// whose element identity is unchanged, so a status flip on the message root
|
||||
// (pending -> complete and back, N rows per stream flush) can no longer
|
||||
// descend into the parts subtree at all. Its props were already the module
|
||||
// constant MESSAGE_PARTS_COMPONENTS, so nothing per-message is captured here.
|
||||
const MESSAGE_PARTS = <MessagePrimitive.Parts components={MESSAGE_PARTS_COMPONENTS} />
|
||||
|
||||
interface MessageActionProps {
|
||||
messageId: string
|
||||
/** Lazy accessor — reads the live message text at action time. Passing the
|
||||
@@ -52,14 +60,12 @@ interface MessageActionProps {
|
||||
onBranchInNewChat?: (messageId: string) => void
|
||||
}
|
||||
|
||||
export const AssistantMessage: FC<{
|
||||
interface AssistantMessageProps {
|
||||
onBranchInNewChat?: (messageId: string) => void
|
||||
onDismissError?: (messageId: string) => void
|
||||
}> = ({ onBranchInNewChat, onDismissError }) => {
|
||||
const messageId = useAuiState(s => s.message.id)
|
||||
const messageRuntime = useMessageRuntime()
|
||||
const { t } = useI18n()
|
||||
}
|
||||
|
||||
export const AssistantMessage: FC<AssistantMessageProps> = props => {
|
||||
// A reply to an inter-agent delivery is part of that exchange, not part of
|
||||
// the human conversation — collapse it under a compact notice ("Reply to
|
||||
// <sender>", expandable), mirroring the sender-side notice the previous
|
||||
@@ -94,14 +100,77 @@ export const AssistantMessage: FC<{
|
||||
return null
|
||||
})
|
||||
|
||||
// PERF: this component must NOT subscribe to the streaming text. Every
|
||||
// selector here returns a value that stays referentially stable across
|
||||
// token flushes (booleans, status strings, '' while running), so the
|
||||
// 30 Hz delta stream only re-renders the markdown part and the tiny
|
||||
// TurnActivityIndicator leaf — not the footer/preview/root subtree.
|
||||
const messageStatus = useAuiState(s => s.message.status?.type)
|
||||
const isRunning = messageStatus === 'running'
|
||||
const isPlaceholder = useAuiState(s => s.message.status?.type === 'running' && s.message.content.length === 0)
|
||||
// The collapse gate below needs the LIVE running status, but only an
|
||||
// inter-agent reply can ever be collapsed. Dispatching on that first keeps
|
||||
// the status subscription out of the standard path entirely — the standard
|
||||
// message root now re-renders for content, never for a pending flip.
|
||||
return interAgentSender ? (
|
||||
<InterAgentAssistantMessage {...props} sender={interAgentSender} />
|
||||
) : (
|
||||
<AssistantMessageBody {...props} />
|
||||
)
|
||||
}
|
||||
|
||||
/** The compact stand-in a settled inter-agent reply collapses to (Grok-bots
|
||||
* parity — the transcript shows the event; the text is one click away). */
|
||||
const InterAgentCollapsedNotice: FC<{ sender: string }> = ({ sender }) => (
|
||||
<div className="flex max-w-[min(86%,44rem)] flex-col gap-0.5 self-center px-2 py-0.5 text-[0.6875rem] leading-5 text-muted-foreground/60">
|
||||
<span className="flex items-center justify-center gap-1.5">
|
||||
<Codicon className="shrink-0 text-muted-foreground/55" name="arrow-small-right" size="0.8125rem" />
|
||||
<span className="wrap-anywhere">Replied to {sender}</span>
|
||||
</span>
|
||||
<details className="self-center">
|
||||
<summary className="cursor-pointer select-none text-center text-muted-foreground/45 hover:text-muted-foreground/70">
|
||||
show reply
|
||||
</summary>
|
||||
<div className="mt-1 max-w-[36rem] rounded-lg border border-(--ui-stroke-tertiary) px-3 py-2 text-left text-[0.75rem] leading-5 text-foreground/85">
|
||||
{MESSAGE_PARTS}
|
||||
</div>
|
||||
</details>
|
||||
</div>
|
||||
)
|
||||
|
||||
/**
|
||||
* An assistant reply that answers an inter-agent delivery. Owns the only
|
||||
* root-level `isRunning` subscription left in this file, and it is confined to
|
||||
* the rare inter-agent case: the reply renders collapsed once it settles, so
|
||||
* the gate genuinely needs live status. Never collapse while streaming — the
|
||||
* user should see progress.
|
||||
*
|
||||
* The collapse is expressed as a CHILD of the normal body, not as a competing
|
||||
* root. Returning a bare MessagePrimitive.Root here for the settled case put a
|
||||
* different element type in this position than the running case
|
||||
* (AssistantMessageBody), so settling unmounted the whole row and mounted a
|
||||
* fresh one — throwing away the DOM the scroll anchor was holding, which can
|
||||
* jump the transcript under the reader. One component, one root, children
|
||||
* vary: settling is now a prop change React applies in place.
|
||||
*/
|
||||
const InterAgentAssistantMessage: FC<AssistantMessageProps & { sender: string }> = ({ sender, ...props }) => {
|
||||
const isRunning = useAuiState(s => s.message.status?.type === 'running')
|
||||
|
||||
return (
|
||||
<AssistantMessageBody
|
||||
{...props}
|
||||
collapsedNotice={isRunning ? null : <InterAgentCollapsedNotice sender={sender} />}
|
||||
/>
|
||||
)
|
||||
}
|
||||
|
||||
const AssistantMessageBody: FC<AssistantMessageProps & { collapsedNotice?: null | ReactNode }> = ({
|
||||
collapsedNotice = null,
|
||||
onBranchInNewChat,
|
||||
onDismissError
|
||||
}) => {
|
||||
const messageId = useAuiState(s => s.message.id)
|
||||
const messageRuntime = useMessageRuntime()
|
||||
const { t } = useI18n()
|
||||
|
||||
// PERF: this component must NOT subscribe to the streaming text, and no
|
||||
// longer subscribes to the streaming STATUS either. Every selector here
|
||||
// returns a value that stays referentially stable across token flushes
|
||||
// (booleans, '' while running), so the 30 Hz delta stream only re-renders
|
||||
// the markdown part and the tiny status leaves — not the footer, the
|
||||
// preview block, or this root.
|
||||
const hasVisibleText = useAuiState(s => contentHasVisibleText(s.message.content))
|
||||
// Sealed mid-turn commentary keeps its text but not the footer, so a
|
||||
// tool-heavy turn doesn't grow a copy/refresh bar per paragraph (see
|
||||
@@ -112,16 +181,154 @@ export const AssistantMessage: FC<{
|
||||
// stable across the 30 Hz delta stream, so this adds no per-token renders).
|
||||
const turnDurationS = useAuiState(s => s.message.metadata?.custom?.durationS as number | undefined)
|
||||
|
||||
// The thinking/stall indicator belongs to the TAIL of the thread, period. A
|
||||
// stale pending bubble mid-transcript (a turn that ended without its settle
|
||||
// event, a steer race) must never show one — a spinner above a later user
|
||||
// message reads as the agent answering out of order. Booleans are stable
|
||||
// across token flushes, so this selector adds no streaming re-renders.
|
||||
const isLastMessage = useAuiState(s => s.thread.messages[s.thread.messages.length - 1]?.id === s.message.id)
|
||||
const getMessageText = useCallback(() => messageContentText(messageRuntime.getState().content), [messageRuntime])
|
||||
|
||||
// Preview targets only materialize once the turn completes — while running
|
||||
// the selector returns '' (stable), so per-token flushes skip the regex
|
||||
// scan and the re-render it would cause.
|
||||
// useEnterAnimation consults `enabled` ONLY when its callback ref fires,
|
||||
// i.e. at mount: the hook parks the value in a ref and returns a
|
||||
// useCallback([]) identity, and its own contract is "`enabled` is captured
|
||||
// at mount-time only — flipping it later doesn't suddenly play the animation
|
||||
// on existing nodes" (see lib/use-enter-animation.ts). So a live
|
||||
// subscription here would re-render this root on every pending flip to feed
|
||||
// a value the hook already ignores. Capture it once, off the runtime, with
|
||||
// no subscription at all.
|
||||
const [initiallyRunning] = useState(() => messageRuntime.getState().status?.type === 'running')
|
||||
const enterRef = useEnterAnimation(initiallyRunning, `assistant-message:${messageId}`)
|
||||
|
||||
// Double-click the reply to heart it (iMessage). Undefined while reactions
|
||||
// are off, so the root carries no listener at all.
|
||||
const onDoubleClick = useTapbackDoubleClick(messageId, 'assistant')
|
||||
|
||||
return (
|
||||
<MessagePrimitive.Root
|
||||
className={cn(
|
||||
'group flex w-full min-w-0 max-w-full flex-col gap-0 self-start overflow-hidden',
|
||||
collapsedNotice && 'pb-(--conversation-turn-gap)'
|
||||
)}
|
||||
data-role="assistant"
|
||||
data-slot="aui_assistant-message-root"
|
||||
// Collapsed inter-agent rows never carried the tapback listener; keeping
|
||||
// that exact truth table means gating it on the notice rather than on
|
||||
// whether the hook returned a handler.
|
||||
onDoubleClick={collapsedNotice ? undefined : onDoubleClick}
|
||||
ref={enterRef}
|
||||
>
|
||||
{collapsedNotice ?? (
|
||||
<>
|
||||
<div
|
||||
className="wrap-anywhere min-w-0 max-w-full overflow-hidden text-pretty text-[length:var(--conversation-text-font-size)] leading-(--dt-line-height) text-foreground"
|
||||
data-slot="aui_assistant-message-content"
|
||||
>
|
||||
{/* Todos render in the composer status stack now, not inline. */}
|
||||
{MESSAGE_PARTS}
|
||||
<AssistantStatusSlot />
|
||||
<AssistantPreviewEmbeds />
|
||||
<MessagePrimitive.Error>
|
||||
<ErrorPrimitive.Root
|
||||
className="mt-1.5 flex items-start gap-1.5 text-[0.78rem] leading-5 text-[color-mix(in_srgb,var(--dt-destructive)_78%,var(--ui-text-secondary))]"
|
||||
role="alert"
|
||||
>
|
||||
<ErrorPrimitive.Message className="min-w-0 flex-1" />
|
||||
{onDismissError && (
|
||||
<TooltipIconButton
|
||||
className="-my-0.5 shrink-0 text-current opacity-70 hover:opacity-100"
|
||||
onClick={() => onDismissError(messageId)}
|
||||
side="top"
|
||||
tooltip={t.assistant.thread.dismissError}
|
||||
>
|
||||
<XIcon className="size-3.5" />
|
||||
</TooltipIconButton>
|
||||
)}
|
||||
</ErrorPrimitive.Root>
|
||||
</MessagePrimitive.Error>
|
||||
</div>
|
||||
<MessageTimelineTimestamp className="px-(--message-text-indent) pt-0.5" suppressIfDuplicatePart />
|
||||
{hasVisibleText && !isInterim && (
|
||||
<AssistantFooter
|
||||
durationS={turnDurationS}
|
||||
getMessageText={getMessageText}
|
||||
messageId={messageId}
|
||||
onBranchInNewChat={onBranchInNewChat}
|
||||
/>
|
||||
)}
|
||||
{/* Last thing in the turn — under the action bar, the way Cursor ends a
|
||||
turn on its summary rather than burying it above the controls. */}
|
||||
<SettledChangedFiles />
|
||||
<StreamingMarker />
|
||||
</>
|
||||
)}
|
||||
</MessagePrimitive.Root>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* PERF leaf: the only subscriber to this message's streaming status inside the
|
||||
* message content. Previously `messageStatus` / `isPlaceholder` /
|
||||
* `isLastMessage` were read by AssistantMessage itself, so every pending flip
|
||||
* re-rendered the whole message subtree — at stream breadth N, N subtrees in a
|
||||
* single commit, which is what widened the recalc scope. Reading them here
|
||||
* confines the flip to this leaf; the sibling parts subtree is a hoisted
|
||||
* constant element and bails out.
|
||||
*
|
||||
* Behaviour is byte-identical to the old inline expression, including the
|
||||
* TAIL-ONLY rule: the activity row belongs to the tail of the thread, period.
|
||||
* A stale pending bubble mid-transcript (a turn that ended without its settle
|
||||
* event, a steer race) must never show one — a spinner above a later user
|
||||
* message reads as the agent answering out of order.
|
||||
*
|
||||
* The activity row is mounted by the TAIL of the thread and decides for itself
|
||||
* whether the turn owes the user a line, so there is deliberately no
|
||||
* `isRunning` gate on the mount here. Gating it on this bubble's own `running`
|
||||
* status was the hole: a turn that seals a bubble mid-flight (message.interim)
|
||||
* or finishes one while the agent keeps going leaves a settled message at the
|
||||
* tail, so the row unmounted and the seconds went uncounted while the
|
||||
* composer's arc border and Stop button said work was still happening.
|
||||
* TurnActivityIndicator subscribes to the status it needs internally, so it is
|
||||
* itself a leaf and this stays off the message root either way.
|
||||
*/
|
||||
const AssistantStatusSlot: FC = () => {
|
||||
// ONE subscription, not one per input. Each useAuiState is a separate store
|
||||
// subscription with its own equality check and its own chance to schedule a
|
||||
// render, and these inputs always move together on a status flip — so
|
||||
// reading them separately just multiplies the wake-ups for a single logical
|
||||
// change. The selector collapses them to one stable string, which bails out
|
||||
// on every flush that does not actually change what this slot renders.
|
||||
const slot = useAuiState(s => {
|
||||
if (s.thread.messages[s.thread.messages.length - 1]?.id !== s.message.id) {
|
||||
return 'none'
|
||||
}
|
||||
|
||||
return s.message.status?.type === 'running' && s.message.content.length === 0 ? 'placeholder' : 'activity'
|
||||
})
|
||||
|
||||
if (slot === 'none') {
|
||||
return null
|
||||
}
|
||||
|
||||
return slot === 'placeholder' ? <ResponseLoadingIndicator /> : <TurnActivityIndicator />
|
||||
}
|
||||
|
||||
/**
|
||||
* PERF leaf: owns the settled-text selector that feeds the link previews.
|
||||
*
|
||||
* This was the last status-dependent read at the message root, and the most
|
||||
* expensive one: the selector flips between '' while running and the full
|
||||
* `messageContentText(content)` join once settled, so every running <-> settled
|
||||
* transition re-ran the join for the whole message AND re-rendered the root.
|
||||
* At stream breadth N that is N joins plus N root re-renders per flip. Reading
|
||||
* it here confines both to this leaf, which renders nothing at all in the
|
||||
* common case.
|
||||
*
|
||||
* The streaming-side optimization is unchanged and still the point of the ''
|
||||
* branch: preview targets only materialize once the turn completes, so while
|
||||
* running the selector returns a stable '' and per-token flushes skip the
|
||||
* regex scan and the re-render it would cause.
|
||||
*
|
||||
* Renders exactly what the root used to render at this position — the same
|
||||
* wrapper div with the same classes, or nothing when there are no targets —
|
||||
* so the DOM is byte-identical either way. A component boundary adds no node
|
||||
* of its own, so unlike StreamingMarker this needs no placement care.
|
||||
*/
|
||||
const AssistantPreviewEmbeds: FC = () => {
|
||||
const completedText = useAuiState(s =>
|
||||
s.message.status?.type === 'running' ? '' : messageContentText(s.message.content)
|
||||
)
|
||||
@@ -134,120 +341,95 @@ export const AssistantMessage: FC<{
|
||||
return pickPrimaryPreviewTarget(extractPreviewTargets(completedText))
|
||||
}, [completedText])
|
||||
|
||||
const getMessageText = useCallback(() => messageContentText(messageRuntime.getState().content), [messageRuntime])
|
||||
if (previewTargets.length === 0) {
|
||||
return null
|
||||
}
|
||||
|
||||
// Cursor's changed-files card only appears once the turn settles: while the
|
||||
// agent is still editing, the tool rows narrate each patch and a card that
|
||||
// grew a row per write would thrash the transcript. `[]` while running keeps
|
||||
// this selector referentially stable across the 30 Hz delta stream.
|
||||
//
|
||||
// It also only rides the LAST turn. The card is a "here's what just landed"
|
||||
// summary, not a per-turn artifact: leaving one behind on every reply would
|
||||
// stack a wall of stale cards down the transcript. Sending the next message
|
||||
// retires it — the working tree it describes is already history by then.
|
||||
return (
|
||||
<div className="mt-3 flex flex-wrap gap-2">
|
||||
{previewTargets.map(target => (
|
||||
<PreviewAttachment key={target} source="explicit-link" target={target} />
|
||||
))}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* PERF leaf: owns the `settledParts` selector so the tail's settle stops
|
||||
* re-rendering the message root. This is the one status-derived selector that
|
||||
* returns an OBJECT (`s.message.parts`) rather than a primitive, so it cannot
|
||||
* bail out on identity churn — keeping it at the root meant every settle
|
||||
* re-rendered the root and everything under it.
|
||||
*
|
||||
* Cursor's changed-files card only appears once the turn settles: while the
|
||||
* agent is still editing, the tool rows narrate each patch and a card that
|
||||
* grew a row per write would thrash the transcript. `EMPTY_PARTS` while
|
||||
* running keeps this selector referentially stable across the 30 Hz delta
|
||||
* stream.
|
||||
*
|
||||
* It also only rides the LAST turn. The card is a "here's what just landed"
|
||||
* summary, not a per-turn artifact: leaving one behind on every reply would
|
||||
* stack a wall of stale cards down the transcript. Sending the next message
|
||||
* retires it — the working tree it describes is already history by then.
|
||||
*/
|
||||
const SettledChangedFiles: FC = () => {
|
||||
const settledParts = useAuiState(s => {
|
||||
const isLastMessage = s.thread.messages[s.thread.messages.length - 1]?.id === s.message.id
|
||||
|
||||
return s.message.status?.type === 'running' || !isLastMessage ? EMPTY_PARTS : s.message.parts
|
||||
})
|
||||
|
||||
const enterRef = useEnterAnimation(isRunning, `assistant-message:${messageId}`)
|
||||
return <ChangedFilesCard parts={settledParts} />
|
||||
}
|
||||
|
||||
// Double-click the reply to heart it (iMessage). Undefined while reactions
|
||||
// are off, so the root carries no listener at all.
|
||||
const onDoubleClick = useTapbackDoubleClick(messageId, 'assistant')
|
||||
|
||||
// Reply inside an inter-agent exchange: render collapsed (Grok-bots
|
||||
// parity — the transcript shows the event; the text is one click away).
|
||||
// Never collapse while streaming: the user should see progress, and the
|
||||
// status selectors above stay live either way.
|
||||
if (interAgentSender && !isRunning) {
|
||||
return (
|
||||
<MessagePrimitive.Root
|
||||
className="group flex w-full min-w-0 max-w-full flex-col gap-0 self-start overflow-hidden pb-(--conversation-turn-gap)"
|
||||
data-role="assistant"
|
||||
data-slot="aui_assistant-message-root"
|
||||
>
|
||||
<div className="flex max-w-[min(86%,44rem)] flex-col gap-0.5 self-center px-2 py-0.5 text-[0.6875rem] leading-5 text-muted-foreground/60">
|
||||
<span className="flex items-center justify-center gap-1.5">
|
||||
<Codicon className="shrink-0 text-muted-foreground/55" name="arrow-small-right" size="0.8125rem" />
|
||||
<span className="wrap-anywhere">Replied to {interAgentSender}</span>
|
||||
</span>
|
||||
<details className="self-center">
|
||||
<summary className="cursor-pointer select-none text-center text-muted-foreground/45 hover:text-muted-foreground/70">
|
||||
show reply
|
||||
</summary>
|
||||
<div className="mt-1 max-w-[36rem] rounded-lg border border-(--ui-stroke-tertiary) px-3 py-2 text-left text-[0.75rem] leading-5 text-foreground/85">
|
||||
<MessagePrimitive.Parts components={MESSAGE_PARTS_COMPONENTS} />
|
||||
</div>
|
||||
</details>
|
||||
</div>
|
||||
</MessagePrimitive.Root>
|
||||
)
|
||||
}
|
||||
/**
|
||||
* Carries the streaming flag that used to sit on the message root as
|
||||
* `data-streaming`.
|
||||
*
|
||||
* The flag has no CSS behind it (every `[data-streaming='true']` rule targets
|
||||
* `[data-slot='code-card']`), but it is not dead: it is the settled-row signal
|
||||
* for the short-session hang repro, which derives the settled count by
|
||||
* subtracting the number of `[data-message-streaming='true']` markers from the
|
||||
* number of message roots, and gates the assistant-response wait on that count
|
||||
* growing. At most one marker per row carries the attribute, which is what
|
||||
* makes the subtraction exact.
|
||||
*
|
||||
* Deliberately NOT named `data-streaming`: shiki-highlighter.tsx puts that
|
||||
* exact attribute on a deferred `[data-slot='code-card']`, which is a
|
||||
* descendant of this root. Once the repro matches on a descendant rather than
|
||||
* the root's own attribute, a shared name would make any message holding a
|
||||
* still-deferred code card read as "still streaming". A distinct name keeps
|
||||
* the signal about the MESSAGE and immune to how deep it sits.
|
||||
*
|
||||
* On the root it was a per-flip attribute write on the element that owns the
|
||||
* whole message subtree, which is the invalidation this prong exists to remove.
|
||||
* Three properties make this placement cheap and behaviour-neutral:
|
||||
*
|
||||
* - A ROOT-LEVEL sibling, not a child of the message content. The
|
||||
* `:first-child` / `:last-child` margin rules in styles.css match blocks
|
||||
* *inside* `[data-slot='aui_assistant-message-content']`; a node added
|
||||
* there would steal `:last-child` from the status indicator and silently
|
||||
* change the gap between bubbles mid-stream. No rule selects message-root
|
||||
* children by position, so this slot is inert.
|
||||
* - PERMANENTLY MOUNTED, toggling only the attribute. Mounting/unmounting per
|
||||
* flip would be a DOM structure change and dirty its siblings; an attribute
|
||||
* write on a childless node invalidates exactly one element.
|
||||
* - `display: none`, so it costs no layout or paint. `querySelectorAll` and
|
||||
* `:has()` still match it — they read the DOM, not the box tree.
|
||||
*
|
||||
* Tracks plain `isRunning` (not tail-only), exactly like the old root
|
||||
* attribute, so the repro's row accounting is unchanged.
|
||||
*/
|
||||
const StreamingMarker: FC = () => {
|
||||
const isRunning = useAuiState(s => s.message.status?.type === 'running')
|
||||
|
||||
return (
|
||||
<MessagePrimitive.Root
|
||||
className="group flex w-full min-w-0 max-w-full flex-col gap-0 self-start overflow-hidden"
|
||||
data-role="assistant"
|
||||
data-slot="aui_assistant-message-root"
|
||||
data-streaming={isRunning ? 'true' : undefined}
|
||||
onDoubleClick={onDoubleClick}
|
||||
ref={enterRef}
|
||||
>
|
||||
<div
|
||||
className="wrap-anywhere min-w-0 max-w-full overflow-hidden text-pretty text-[length:var(--conversation-text-font-size)] leading-(--dt-line-height) text-foreground"
|
||||
data-slot="aui_assistant-message-content"
|
||||
>
|
||||
{/* Todos render in the composer status stack now, not inline. */}
|
||||
<MessagePrimitive.Parts components={MESSAGE_PARTS_COMPONENTS} />
|
||||
{/* The activity row is mounted by the TAIL of the thread and decides
|
||||
for itself whether the turn owes the user a line. Gating the mount
|
||||
on this bubble's own `running` status was the hole: a turn that
|
||||
seals a bubble mid-flight (message.interim) or finishes one while
|
||||
the agent keeps going leaves a settled message at the tail, so the
|
||||
row unmounted and the seconds went uncounted while the composer's
|
||||
arc border and Stop button said work was still happening. */}
|
||||
{isLastMessage && (isPlaceholder ? <ResponseLoadingIndicator /> : <TurnActivityIndicator />)}
|
||||
{previewTargets.length > 0 && (
|
||||
<div className="mt-3 flex flex-wrap gap-2">
|
||||
{previewTargets.map(target => (
|
||||
<PreviewAttachment key={target} source="explicit-link" target={target} />
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
<MessagePrimitive.Error>
|
||||
<ErrorPrimitive.Root
|
||||
className="mt-1.5 flex items-start gap-1.5 text-[0.78rem] leading-5 text-[color-mix(in_srgb,var(--dt-destructive)_78%,var(--ui-text-secondary))]"
|
||||
role="alert"
|
||||
>
|
||||
<ErrorPrimitive.Message className="min-w-0 flex-1" />
|
||||
{onDismissError && (
|
||||
<TooltipIconButton
|
||||
className="-my-0.5 shrink-0 text-current opacity-70 hover:opacity-100"
|
||||
onClick={() => onDismissError(messageId)}
|
||||
side="top"
|
||||
tooltip={t.assistant.thread.dismissError}
|
||||
>
|
||||
<XIcon className="size-3.5" />
|
||||
</TooltipIconButton>
|
||||
)}
|
||||
</ErrorPrimitive.Root>
|
||||
</MessagePrimitive.Error>
|
||||
</div>
|
||||
<MessageTimelineTimestamp className="px-(--message-text-indent) pt-0.5" suppressIfDuplicatePart />
|
||||
{hasVisibleText && !isInterim && (
|
||||
<AssistantFooter
|
||||
durationS={turnDurationS}
|
||||
getMessageText={getMessageText}
|
||||
messageId={messageId}
|
||||
onBranchInNewChat={onBranchInNewChat}
|
||||
/>
|
||||
)}
|
||||
{/* Last thing in the turn — under the action bar, the way Cursor ends a
|
||||
turn on its summary rather than burying it above the controls. */}
|
||||
<ChangedFilesCard parts={settledParts} />
|
||||
</MessagePrimitive.Root>
|
||||
<span
|
||||
aria-hidden="true"
|
||||
className="hidden"
|
||||
data-message-streaming={isRunning ? 'true' : undefined}
|
||||
data-slot="aui_message-streaming-marker"
|
||||
/>
|
||||
)
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,119 @@
|
||||
// Two contracts with no coverage before the invalidation-scoping work split
|
||||
// AssistantMessage into InterAgentAssistantMessage + AssistantMessageBody:
|
||||
//
|
||||
// 1. The collapse gate. A reply to an inter-agent delivery renders collapsed
|
||||
// ("Replied to <sender>", expandable) ONLY once it settles — never while it
|
||||
// streams, because the user should see progress. That gate is the sole
|
||||
// remaining root-level `isRunning` subscription, so it is the thing most
|
||||
// likely to break if the split is revisited.
|
||||
// 2. The streaming marker. `data-message-streaming` moved off the message root
|
||||
// onto a permanently-mounted hidden leaf, and
|
||||
// scripts/run-short-session-hang-repro.mjs derives its settled-row count by
|
||||
// subtracting `[data-message-streaming="true"]` markers from message roots.
|
||||
// Nothing in the app itself reads it, so without this test a delete would
|
||||
// look free and would silently regress that repro's response gate.
|
||||
import { AssistantRuntimeProvider, type ThreadMessage, useExternalStoreRuntime } from '@assistant-ui/react'
|
||||
import { cleanup, render, screen } from '@testing-library/react'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { Thread } from '.'
|
||||
|
||||
const createdAt = new Date('2026-05-01T00:00:00.000Z')
|
||||
|
||||
class TestResizeObserver {
|
||||
observe() {}
|
||||
unobserve() {}
|
||||
disconnect() {}
|
||||
}
|
||||
vi.stubGlobal('ResizeObserver', TestResizeObserver)
|
||||
vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) =>
|
||||
window.setTimeout(() => callback(performance.now()), 0)
|
||||
)
|
||||
vi.stubGlobal('cancelAnimationFrame', (id: number) => window.clearTimeout(id))
|
||||
vi.stubGlobal('CSS', { escape: (str: string) => str })
|
||||
|
||||
Element.prototype.scrollTo = function scrollTo() {}
|
||||
|
||||
Element.prototype.animate = function animate() {
|
||||
return { cancel() {}, finished: Promise.resolve() } as unknown as Animation
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
})
|
||||
|
||||
const assistantMetadata = { unstable_state: null, unstable_annotations: [], unstable_data: [], steps: [], custom: {} }
|
||||
|
||||
function user(id: string, text: string): ThreadMessage {
|
||||
return {
|
||||
id,
|
||||
role: 'user',
|
||||
content: [{ type: 'text', text }],
|
||||
attachments: [],
|
||||
createdAt,
|
||||
metadata: { custom: {} }
|
||||
} as ThreadMessage
|
||||
}
|
||||
|
||||
function assistant(id: string, text: string, running: boolean): ThreadMessage {
|
||||
return {
|
||||
id,
|
||||
role: 'assistant',
|
||||
content: text ? [{ type: 'text', text }] : [],
|
||||
status: running ? { type: 'running' } : { type: 'complete', reason: 'stop' },
|
||||
createdAt,
|
||||
metadata: assistantMetadata
|
||||
} as ThreadMessage
|
||||
}
|
||||
|
||||
function Harness({ messages }: { messages: ThreadMessage[] }) {
|
||||
const runtime = useExternalStoreRuntime<ThreadMessage>({
|
||||
messages,
|
||||
isRunning: messages.at(-1)?.status?.type === 'running',
|
||||
onNew: async () => {}
|
||||
})
|
||||
|
||||
return (
|
||||
<AssistantRuntimeProvider runtime={runtime}>
|
||||
<Thread />
|
||||
</AssistantRuntimeProvider>
|
||||
)
|
||||
}
|
||||
|
||||
const DELIVERY = 'Message from 🤖 Hermes (@hermes): please check the build'
|
||||
|
||||
describe('inter-agent collapse gate', () => {
|
||||
it('collapses a settled reply to an inter-agent delivery', async () => {
|
||||
render(<Harness messages={[user('u1', DELIVERY), assistant('a1', 'build is green', false)]} />)
|
||||
|
||||
expect(await screen.findByText(/Replied to/)).toBeTruthy()
|
||||
expect(screen.getByText('show reply')).toBeTruthy()
|
||||
})
|
||||
|
||||
it('does NOT collapse while that reply is still streaming', async () => {
|
||||
const { container } = render(<Harness messages={[user('u1', DELIVERY), assistant('a1', 'working on it', true)]} />)
|
||||
|
||||
await screen.findByText('working on it')
|
||||
expect(screen.queryByText('show reply')).toBeNull()
|
||||
// Expanded => the full body root, which carries the streaming marker.
|
||||
expect(container.querySelector('[data-message-streaming="true"]')).toBeTruthy()
|
||||
})
|
||||
|
||||
it('leaves an ordinary reply expanded', async () => {
|
||||
render(<Harness messages={[user('u1', 'ordinary question'), assistant('a1', 'ordinary answer', false)]} />)
|
||||
|
||||
await screen.findByText('ordinary answer')
|
||||
expect(screen.queryByText(/Replied to/)).toBeNull()
|
||||
})
|
||||
|
||||
it('clears the streaming marker once the turn settles', async () => {
|
||||
const { container } = render(<Harness messages={[user('u1', 'q'), assistant('a1', 'done', false)]} />)
|
||||
|
||||
await screen.findByText('done')
|
||||
expect(container.querySelector('[data-message-streaming="true"]')).toBeNull()
|
||||
// The marker element itself stays mounted (attribute toggles, no remount).
|
||||
expect(
|
||||
container.querySelector('[data-slot="aui_assistant-message-root"] [data-slot="aui_message-streaming-marker"]')
|
||||
).toBeTruthy()
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,100 @@
|
||||
// Link previews moved off the message root into the AssistantPreviewEmbeds
|
||||
// leaf, because the selector behind them (`'' while running`, the full
|
||||
// `messageContentText(content)` join once settled) flipped on every
|
||||
// running <-> settled transition and re-rendered the root with it.
|
||||
//
|
||||
// Two things are pinned here. That the embed still renders at all — the move
|
||||
// was verbatim JSX and had no coverage before. And that it renders ONLY once
|
||||
// the turn settles: the '' branch is a deliberate streaming optimization (it
|
||||
// keeps the selector referentially stable so per-token flushes skip the regex
|
||||
// scan), so a rewrite that drops it would make previews flicker in mid-stream
|
||||
// with nothing to catch it.
|
||||
import { AssistantRuntimeProvider, type ThreadMessage, useExternalStoreRuntime } from '@assistant-ui/react'
|
||||
import { cleanup, render, screen } from '@testing-library/react'
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { Thread } from '.'
|
||||
|
||||
const createdAt = new Date('2026-05-01T00:00:00.000Z')
|
||||
|
||||
class TestResizeObserver {
|
||||
observe() {}
|
||||
unobserve() {}
|
||||
disconnect() {}
|
||||
}
|
||||
|
||||
vi.stubGlobal('ResizeObserver', TestResizeObserver)
|
||||
vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) =>
|
||||
window.setTimeout(() => callback(performance.now()), 0)
|
||||
)
|
||||
vi.stubGlobal('cancelAnimationFrame', (id: number) => window.clearTimeout(id))
|
||||
vi.stubGlobal('CSS', { escape: (str: string) => str })
|
||||
|
||||
Element.prototype.scrollTo = function scrollTo() {}
|
||||
|
||||
Element.prototype.animate = function animate() {
|
||||
return { cancel() {}, finished: Promise.resolve() } as unknown as Animation
|
||||
}
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
})
|
||||
|
||||
const assistantMetadata = { unstable_state: null, unstable_annotations: [], unstable_data: [], steps: [], custom: {} }
|
||||
|
||||
function user(id: string, text: string): ThreadMessage {
|
||||
return {
|
||||
id,
|
||||
role: 'user',
|
||||
content: [{ type: 'text', text }],
|
||||
attachments: [],
|
||||
createdAt,
|
||||
metadata: { custom: {} }
|
||||
} as ThreadMessage
|
||||
}
|
||||
|
||||
function assistant(id: string, text: string, running: boolean): ThreadMessage {
|
||||
return {
|
||||
id,
|
||||
role: 'assistant',
|
||||
content: text ? [{ type: 'text', text }] : [],
|
||||
status: running ? { type: 'running' } : { type: 'complete', reason: 'stop' },
|
||||
createdAt,
|
||||
metadata: assistantMetadata
|
||||
} as ThreadMessage
|
||||
}
|
||||
|
||||
function Harness({ messages }: { messages: ThreadMessage[] }) {
|
||||
const runtime = useExternalStoreRuntime<ThreadMessage>({
|
||||
messages,
|
||||
isRunning: messages.at(-1)?.status?.type === 'running',
|
||||
onNew: async () => {}
|
||||
})
|
||||
|
||||
return (
|
||||
<AssistantRuntimeProvider runtime={runtime}>
|
||||
<Thread />
|
||||
</AssistantRuntimeProvider>
|
||||
)
|
||||
}
|
||||
|
||||
const TARGET = 'https://example.com/docs'
|
||||
const WITH_PREVIEW = `Serving now: [Preview: example](#preview/${TARGET})`
|
||||
|
||||
describe('settled-turn link previews', () => {
|
||||
it('renders the embed once the turn has settled', async () => {
|
||||
const { container } = render(<Harness messages={[user('u1', 'start it'), assistant('a1', WITH_PREVIEW, false)]} />)
|
||||
|
||||
await screen.findByText('Serving now:', { exact: false })
|
||||
|
||||
expect(container.querySelector(`[title="${TARGET}"]`)).toBeTruthy()
|
||||
})
|
||||
|
||||
it('does not render the embed while the turn is still running', async () => {
|
||||
const { container } = render(<Harness messages={[user('u1', 'start it'), assistant('a1', WITH_PREVIEW, true)]} />)
|
||||
|
||||
await screen.findByText('Serving now:', { exact: false })
|
||||
|
||||
expect(container.querySelector(`[title="${TARGET}"]`)).toBeNull()
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,166 @@
|
||||
// The invalidation-scoping property, as a render-count contract.
|
||||
//
|
||||
// A streaming turn flips its message status many times a second, and at
|
||||
// stream breadth N that is N status flips per flush. The whole point of this
|
||||
// work is that a flip re-renders only the small leaves that actually display
|
||||
// status — never the message ROOT, whose subtree is the entire rendered
|
||||
// message and whose re-render is what widened style recalculation to document
|
||||
// scope.
|
||||
//
|
||||
// That property is invisible to a DOM assertion: the transcript looks
|
||||
// identical either way. So this counts renders instead. AssistantMessageBody
|
||||
// is the root component, and `useTapbackDoubleClick` is called by it and by
|
||||
// nothing else in the tree, which makes it an exact render counter for the
|
||||
// root without needing to export or wrap an internal component.
|
||||
import { AssistantRuntimeProvider, type ThreadMessage, useExternalStoreRuntime } from '@assistant-ui/react'
|
||||
import { cleanup, render, waitFor } from '@testing-library/react'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import type * as messageReactionsModule from '@/components/assistant-ui/thread/use-message-reactions'
|
||||
|
||||
import { Thread } from '.'
|
||||
|
||||
let rootRenders = 0
|
||||
|
||||
vi.mock('@/components/assistant-ui/thread/use-message-reactions', async importActual => {
|
||||
const actual = await importActual<typeof messageReactionsModule>()
|
||||
|
||||
return {
|
||||
...actual,
|
||||
useTapbackDoubleClick: (messageId: string, role: 'assistant' | 'user') => {
|
||||
if (role === 'assistant') {
|
||||
rootRenders += 1
|
||||
}
|
||||
|
||||
return actual.useTapbackDoubleClick(messageId, role)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
const createdAt = new Date('2026-05-01T00:00:00.000Z')
|
||||
|
||||
class TestResizeObserver {
|
||||
observe() {}
|
||||
unobserve() {}
|
||||
disconnect() {}
|
||||
}
|
||||
vi.stubGlobal('ResizeObserver', TestResizeObserver)
|
||||
vi.stubGlobal('requestAnimationFrame', (callback: FrameRequestCallback) =>
|
||||
window.setTimeout(() => callback(performance.now()), 0)
|
||||
)
|
||||
vi.stubGlobal('cancelAnimationFrame', (id: number) => window.clearTimeout(id))
|
||||
vi.stubGlobal('CSS', { escape: (str: string) => str })
|
||||
|
||||
Element.prototype.scrollTo = function scrollTo() {}
|
||||
|
||||
Element.prototype.animate = function animate() {
|
||||
return { cancel() {}, finished: Promise.resolve() } as unknown as Animation
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
rootRenders = 0
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
cleanup()
|
||||
})
|
||||
|
||||
const assistantMetadata = { unstable_state: null, unstable_annotations: [], unstable_data: [], steps: [], custom: {} }
|
||||
|
||||
function user(id: string, text: string): ThreadMessage {
|
||||
return {
|
||||
id,
|
||||
role: 'user',
|
||||
content: [{ type: 'text', text }],
|
||||
attachments: [],
|
||||
createdAt,
|
||||
metadata: { custom: {} }
|
||||
} as ThreadMessage
|
||||
}
|
||||
|
||||
function assistant(id: string, text: string, running: boolean): ThreadMessage {
|
||||
return {
|
||||
id,
|
||||
role: 'assistant',
|
||||
content: text ? [{ type: 'text', text }] : [],
|
||||
status: running ? { type: 'running' } : { type: 'complete', reason: 'stop' },
|
||||
createdAt,
|
||||
metadata: assistantMetadata
|
||||
} as ThreadMessage
|
||||
}
|
||||
|
||||
function Harness({ messages }: { messages: ThreadMessage[] }) {
|
||||
const runtime = useExternalStoreRuntime<ThreadMessage>({
|
||||
messages,
|
||||
isRunning: messages.at(-1)?.status?.type === 'running',
|
||||
onNew: async () => {}
|
||||
})
|
||||
|
||||
return (
|
||||
<AssistantRuntimeProvider runtime={runtime}>
|
||||
<Thread />
|
||||
</AssistantRuntimeProvider>
|
||||
)
|
||||
}
|
||||
|
||||
describe('streaming-status invalidation scope', () => {
|
||||
it('does not re-render the message root when the turn settles', async () => {
|
||||
const messages = [user('u1', 'question'), assistant('a1', 'partial answer', true)]
|
||||
const { container, findByText, rerender } = render(<Harness messages={messages} />)
|
||||
|
||||
await findByText('partial answer')
|
||||
// The leaf carries the streaming flag while the turn is in flight.
|
||||
expect(container.querySelector('[data-message-streaming="true"]')).toBeTruthy()
|
||||
|
||||
const rendersWhileStreaming = rootRenders
|
||||
|
||||
rerender(<Harness messages={[messages[0], assistant('a1', 'partial answer', false)]} />)
|
||||
|
||||
// The leaf saw the flip...
|
||||
await waitFor(() => {
|
||||
expect(container.querySelector('[data-message-streaming="true"]')).toBeNull()
|
||||
})
|
||||
// ...on the same permanently-mounted node (attribute toggle, no remount)...
|
||||
expect(container.querySelector('[data-slot="aui_message-streaming-marker"]')).toBeTruthy()
|
||||
// ...and the root did not re-render for it.
|
||||
expect(rootRenders).toBe(rendersWhileStreaming)
|
||||
})
|
||||
|
||||
it('does not re-render the message root when streaming text arrives', async () => {
|
||||
const messages = [user('u1', 'question'), assistant('a1', 'one', true)]
|
||||
const { findByText, rerender } = render(<Harness messages={messages} />)
|
||||
|
||||
await findByText('one')
|
||||
const rendersAfterFirstToken = rootRenders
|
||||
|
||||
// A delta flush: same status, more text. The root must not subscribe to
|
||||
// the streaming text either — only the markdown part re-renders.
|
||||
rerender(<Harness messages={[messages[0], assistant('a1', 'one two', true)]} />)
|
||||
await findByText('one two')
|
||||
|
||||
expect(rootRenders).toBe(rendersAfterFirstToken)
|
||||
})
|
||||
|
||||
it('keeps the same root DOM node across the settle transition', async () => {
|
||||
// The inter-agent reply collapses once it settles. Rendering that as a
|
||||
// competing root swapped the element type at this position, so React
|
||||
// unmounted the row and mounted a fresh one — discarding the DOM the
|
||||
// scroll anchor was holding.
|
||||
const delivery = 'Message from 🤖 Hermes (@hermes): please check the build'
|
||||
const messages = [user('u1', delivery), assistant('a1', 'working on it', true)]
|
||||
const { container, findByText, rerender } = render(<Harness messages={messages} />)
|
||||
|
||||
await findByText('working on it')
|
||||
const before = container.querySelector('[data-slot="aui_assistant-message-root"]')
|
||||
|
||||
expect(before).toBeTruthy()
|
||||
|
||||
rerender(<Harness messages={[messages[0], assistant('a1', 'working on it', false)]} />)
|
||||
await findByText(/Replied to/)
|
||||
|
||||
const after = container.querySelector('[data-slot="aui_assistant-message-root"]')
|
||||
|
||||
// Same element, updated in place — not a remount.
|
||||
expect(after).toBe(before)
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,123 @@
|
||||
/* Compositor-only glyph spinner.
|
||||
*
|
||||
* The previous implementation ticked a setInterval and replaced
|
||||
* `glyph.textContent` every frame. Replacing a text node is a structural DOM
|
||||
* mutation, so each tick scheduled a style recalculation — and because these
|
||||
* spinners sit inside the transcript, that recalc resolved against the whole
|
||||
* document. Scheduler attribution on an incident trace put the large majority
|
||||
* of the wide document-scale recalcs on this ticker, and none of the cheap
|
||||
* scoped ones; the trace numbers are in the PR that introduced this file.
|
||||
*
|
||||
* Every frame is now in the DOM from mount as a vertical strip, and a
|
||||
* `transform: translateY` keyframes animation scrolls it one frame at a time.
|
||||
* Transform animations run on the compositor: no JS timer, no text mutation,
|
||||
* no style recalc, no layout, nothing scheduled per frame.
|
||||
*
|
||||
* The strip is N frames tall and each frame is exactly 1em, so travelling N
|
||||
* frame-heights scrolls through the whole strip exactly once. `steps(N)`
|
||||
* (jump-end) samples that travel at 0, 1/N, ... (N-1)/N, i.e. it parks on
|
||||
* frame 0..N-1 for one interval each and then wraps — the same sequence and
|
||||
* cadence the ticker produced. Both N and the duration arrive as custom
|
||||
* properties set inline per spinner, so all 18 spinner variants (16 distinct
|
||||
* frame/interval shapes) share this one rule set with no generated or
|
||||
* colliding per-variant keyframes. Every var() carries the braille default as
|
||||
* its fallback, so a missing custom property degrades to a working spinner
|
||||
* rather than an invalid declaration.
|
||||
*/
|
||||
|
||||
.glyph-spinner {
|
||||
/* Single source of truth for the frame metric: the clip viewport, each frame
|
||||
box, and the keyframe travel all derive from this, so they cannot drift
|
||||
apart. `em` resolves against the element it is used on, and font-size is
|
||||
uniform across this subtree — nothing below .glyph-spinner declares one —
|
||||
so the strip's `em` and the frame's `em` are the same length. */
|
||||
--glyph-spinner-frame-height: 1em;
|
||||
|
||||
display: block;
|
||||
height: var(--glyph-spinner-frame-height, 1em);
|
||||
overflow: hidden;
|
||||
line-height: 1;
|
||||
}
|
||||
|
||||
/* Decorative and aria-hidden, so it must not land in a transcript copy. These
|
||||
spinners sit inside [data-selectable-text] subtrees (tool/delegate,
|
||||
tool/fallback), where the strip would otherwise contribute all N glyphs to a
|
||||
selection where the old one-character implementation contributed one.
|
||||
Descendants are named explicitly rather than left to inherit: the competing
|
||||
`[data-selectable-text='true'] *` rule carries the same (0,1,0) specificity,
|
||||
so relying on inheritance would make the winner depend on stylesheet order. */
|
||||
.glyph-spinner,
|
||||
.glyph-spinner *,
|
||||
[data-selectable-text='true'] .glyph-spinner,
|
||||
[data-selectable-text='true'] .glyph-spinner * {
|
||||
-webkit-user-select: none;
|
||||
user-select: none;
|
||||
}
|
||||
|
||||
.glyph-spinner__strip {
|
||||
display: block;
|
||||
animation: glyph-spinner-advance var(--glyph-spinner-duration, 800ms) steps(var(--glyph-spinner-frames, 10)) infinite;
|
||||
}
|
||||
|
||||
/* Promote to a compositor layer — without it, trace instrumentation recorded a
|
||||
non-zero compositeFailed on every animation record. Scoped to spinners that
|
||||
are actually animating: a permanently promoted layer per parked spinner is
|
||||
pure memory at fan-out breadth, where many spinners sit mounted and paused
|
||||
at once. */
|
||||
.glyph-spinner:not([data-paused='true']) .glyph-spinner__strip {
|
||||
will-change: transform;
|
||||
}
|
||||
|
||||
/* The global renderer pause and the reduced-motion freeze stop the animation;
|
||||
drop the layer hint there too, so parked spinners hold no compositor layer. */
|
||||
:root[data-renderer-animations-paused] .glyph-spinner__strip {
|
||||
will-change: auto;
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
.glyph-spinner__strip {
|
||||
will-change: auto;
|
||||
}
|
||||
}
|
||||
|
||||
.glyph-spinner__frame {
|
||||
display: block;
|
||||
height: var(--glyph-spinner-frame-height, 1em);
|
||||
/* Clip each frame to its own box. Braille (U+2800..U+28FF) renders from a
|
||||
system fallback face — JetBrains Mono has no glyphs in that block — and
|
||||
fallback metrics are not guaranteed to fit the 1em box, so ink from the
|
||||
neighbouring frames could otherwise bleed into the viewport. */
|
||||
overflow: hidden;
|
||||
line-height: 1;
|
||||
}
|
||||
|
||||
/* Parked, for either of two reasons: the pane is a kept-alive but inactive tab
|
||||
(usePaneVisible), or the caller passed `paused` to hold the spinner still
|
||||
while it stays mounted through a fade-out (ChatSwapOverlay). Hidden panes
|
||||
normally get `content-visibility: hidden`, which already stops animations in
|
||||
the skipped subtree, but that containment has a runtime kill switch and
|
||||
older pane layers only set `visibility: hidden` — which does NOT stop an
|
||||
animation. This keeps the original explicit guarantee ("N mounted tabs each
|
||||
ticking burns CPU for pixels nobody can see") independent of which of those
|
||||
is in play. */
|
||||
.glyph-spinner[data-paused='true'] .glyph-spinner__strip {
|
||||
animation-play-state: paused;
|
||||
}
|
||||
|
||||
/* The travel MUST be an absolute length, never a percentage. A percentage
|
||||
translate resolves against the strip's own box, so Chromium treats the
|
||||
animation as layout-dependent and refuses to composite it — trace
|
||||
instrumentation showed compositeFailed set on every record while the
|
||||
keyframes used translateY(-100%), and clean compositing on a sibling
|
||||
transform animation that travelled an absolute length (numbers in the PR).
|
||||
N frames x the frame height is the same distance, computed rather than
|
||||
box-relative. */
|
||||
@keyframes glyph-spinner-advance {
|
||||
from {
|
||||
transform: translateY(0);
|
||||
}
|
||||
|
||||
to {
|
||||
transform: translateY(calc(var(--glyph-spinner-frames, 10) * -1 * var(--glyph-spinner-frame-height, 1em)));
|
||||
}
|
||||
}
|
||||
@@ -1,15 +1,49 @@
|
||||
import { act, render, screen } from '@testing-library/react'
|
||||
// GlyphSpinner animates on the compositor: every frame is in the DOM from
|
||||
// mount and a transform keyframes animation scrolls between them. It has no
|
||||
// timer and performs no per-frame DOM write, because the setInterval +
|
||||
// `glyph.textContent` ticker this replaced was scheduling a document-scale
|
||||
// style recalculation on every tick.
|
||||
//
|
||||
// These tests therefore pin the DATA and WIRING that make the CSS correct,
|
||||
// which is all jsdom can see — it has no animation engine, so the motion
|
||||
// itself is not observable here:
|
||||
//
|
||||
// - the strip carries every frame, in order, so `steps(N)` lands on each one;
|
||||
// - the custom properties feeding `steps()` / duration match the source data,
|
||||
// so cadence per variant is preserved;
|
||||
// - no timer is ever created (the ticker is really gone);
|
||||
// - the kept-alive-tab gate still resolves to a paused animation.
|
||||
//
|
||||
// The CSS itself is verified where it can actually run: e2e/glyph-spinner.spec.ts
|
||||
// reads getComputedStyle / getAnimations in a real browser. Asserting on the
|
||||
// text of the stylesheet from here would test the shape of the source rather
|
||||
// than its behaviour (banned outright — see AGENTS.md), and did in fact break
|
||||
// on a pure var()-fallback edit that changed no rendered pixel.
|
||||
import { render, screen } from '@testing-library/react'
|
||||
import { Profiler, type ProfilerOnRenderCallback } from 'react'
|
||||
import spinners from 'unicode-animations'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import { PaneVisibleContext } from '@/components/pane-shell/pane-visibility'
|
||||
|
||||
import { GlyphSpinner } from './glyph-spinner'
|
||||
|
||||
const BRAILLE = spinners.braille
|
||||
|
||||
function strip(): HTMLElement {
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
const found = status.querySelector<HTMLElement>('.glyph-spinner__strip')
|
||||
|
||||
if (!found) {
|
||||
throw new Error('no frame strip rendered')
|
||||
}
|
||||
|
||||
return found
|
||||
}
|
||||
|
||||
describe('GlyphSpinner', () => {
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers()
|
||||
vi.spyOn(globalThis.document, 'hasFocus').mockReturnValue(true)
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
@@ -18,7 +52,41 @@ describe('GlyphSpinner', () => {
|
||||
vi.useRealTimers()
|
||||
})
|
||||
|
||||
it('advances its glyph without an update-phase React commit', () => {
|
||||
it('renders every frame in source order as the scroll strip', () => {
|
||||
render(<GlyphSpinner spinner="braille" />)
|
||||
|
||||
const frames = [...strip().querySelectorAll('.glyph-spinner__frame')].map(node => node.textContent)
|
||||
|
||||
expect(frames).toEqual([...BRAILLE.frames])
|
||||
// The old ticker started on frame 0; steps() starts the strip there too.
|
||||
expect(frames[0]).toBe('⠋')
|
||||
})
|
||||
|
||||
it('feeds steps() and the duration from the spinner data, so cadence is unchanged', () => {
|
||||
render(<GlyphSpinner spinner="braille" />)
|
||||
|
||||
const style = strip().style
|
||||
|
||||
// One full cycle is frames x interval; steps(frames) parks on each frame
|
||||
// for exactly `interval` ms, which is what the setInterval did.
|
||||
expect(style.getPropertyValue('--glyph-spinner-frames')).toBe(String(BRAILLE.frames.length))
|
||||
expect(style.getPropertyValue('--glyph-spinner-duration')).toBe(`${BRAILLE.frames.length * BRAILLE.interval}ms`)
|
||||
})
|
||||
|
||||
it('keeps per-variant cadence distinct', () => {
|
||||
render(<GlyphSpinner ariaLabel="Working" spinner="breathe" />)
|
||||
|
||||
const node = screen.getByRole('status', { name: 'Working' })
|
||||
const found = node.querySelector<HTMLElement>('.glyph-spinner__strip')!
|
||||
const breathe = spinners.breathe
|
||||
|
||||
expect(found.querySelectorAll('.glyph-spinner__frame')).toHaveLength(breathe.frames.length)
|
||||
expect(found.style.getPropertyValue('--glyph-spinner-duration')).toBe(
|
||||
`${breathe.frames.length * breathe.interval}ms`
|
||||
)
|
||||
})
|
||||
|
||||
it('creates no timer and no update-phase React commit', () => {
|
||||
let updateCommits = 0
|
||||
|
||||
const onRender: ProfilerOnRenderCallback = (_id, phase) => {
|
||||
@@ -33,135 +101,55 @@ describe('GlyphSpinner', () => {
|
||||
</Profiler>
|
||||
)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
expect(status.textContent).toBe('⠋')
|
||||
// The whole point: the ticker is gone, so nothing is scheduled at all.
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
vi.advanceTimersByTime(5_000)
|
||||
|
||||
expect(status.textContent).toBe('⠙')
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
expect(updateCommits).toBe(0)
|
||||
})
|
||||
|
||||
it('does not tick while its kept-alive pane is hidden', () => {
|
||||
it('pauses while its kept-alive pane is hidden, and resumes when shown', () => {
|
||||
const { rerender } = render(
|
||||
<PaneVisibleContext.Provider value={false}>
|
||||
<GlyphSpinner spinner="braille" />
|
||||
</PaneVisibleContext.Provider>
|
||||
)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
const viewport = () => screen.getByRole('status', { name: 'Loading' }).querySelector('.glyph-spinner')
|
||||
|
||||
expect(status.textContent).toBe('⠋')
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
expect(viewport()?.getAttribute('data-paused')).toBe('true')
|
||||
|
||||
rerender(
|
||||
<PaneVisibleContext.Provider value>
|
||||
<GlyphSpinner spinner="braille" />
|
||||
</PaneVisibleContext.Provider>
|
||||
)
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
expect(status.textContent).toBe('⠙')
|
||||
|
||||
rerender(
|
||||
<PaneVisibleContext.Provider value={false}>
|
||||
<GlyphSpinner spinner="braille" />
|
||||
</PaneVisibleContext.Provider>
|
||||
)
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
const frozen = status.textContent
|
||||
act(() => vi.advanceTimersByTime(800))
|
||||
expect(status.textContent).toBe(frozen)
|
||||
expect(viewport()?.hasAttribute('data-paused')).toBe(false)
|
||||
})
|
||||
|
||||
it('suspends animation while the Desktop window is inactive', () => {
|
||||
it('hides the decorative frames from assistive tech', () => {
|
||||
render(<GlyphSpinner spinner="braille" />)
|
||||
|
||||
// role="status" is a live region. The frames must not be announced — the
|
||||
// old implementation rewrote this region's text ~12x/second.
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => window.dispatchEvent(new Event('blur')))
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
const frozen = status.textContent
|
||||
act(() => vi.advanceTimersByTime(800))
|
||||
expect(status.textContent).toBe(frozen)
|
||||
|
||||
act(() => window.dispatchEvent(new Event('focus')))
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
expect(status.textContent).not.toBe(frozen)
|
||||
expect(status.querySelector('.glyph-spinner')?.getAttribute('aria-hidden')).toBe('true')
|
||||
})
|
||||
|
||||
it('suspends animation while the Electron window is minimized or hidden, then resumes on restore', () => {
|
||||
let windowStateCallback: ((payload: { isMinimized?: boolean; isVisible?: boolean }) => void) | null = null
|
||||
it('pauses on request while staying mounted', () => {
|
||||
// For a caller that keeps the spinner in the tree through a fade-out
|
||||
// (ChatSwapOverlay) and does not want it animating once the wait is over.
|
||||
const { rerender } = render(<GlyphSpinner paused spinner="braille" />)
|
||||
const viewport = () => screen.getByRole('status', { name: 'Loading' }).querySelector('.glyph-spinner')
|
||||
|
||||
Object.defineProperty(window, 'hermesDesktop', {
|
||||
configurable: true,
|
||||
value: {
|
||||
onWindowStateChanged: vi.fn((callback: typeof windowStateCallback) => {
|
||||
windowStateCallback = callback
|
||||
expect(viewport()?.getAttribute('data-paused')).toBe('true')
|
||||
|
||||
return () => {
|
||||
if (windowStateCallback === callback) {
|
||||
windowStateCallback = null
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
})
|
||||
rerender(<GlyphSpinner paused={false} spinner="braille" />)
|
||||
|
||||
try {
|
||||
render(<GlyphSpinner spinner="braille" />)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
expect(windowStateCallback).not.toBeNull()
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => windowStateCallback?.({ isMinimized: true, isVisible: false }))
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
const frozen = status.textContent
|
||||
act(() => vi.advanceTimersByTime(800))
|
||||
expect(status.textContent).toBe(frozen)
|
||||
|
||||
act(() => windowStateCallback?.({ isMinimized: false, isVisible: true }))
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
expect(status.textContent).not.toBe(frozen)
|
||||
} finally {
|
||||
delete (window as unknown as { hermesDesktop?: unknown }).hermesDesktop
|
||||
}
|
||||
})
|
||||
|
||||
it('suspends animation while the document is hidden', () => {
|
||||
render(<GlyphSpinner spinner="braille" />)
|
||||
|
||||
const status = screen.getByRole('status', { name: 'Loading' })
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
Object.defineProperty(document, 'visibilityState', { configurable: true, value: 'hidden' })
|
||||
|
||||
try {
|
||||
act(() => document.dispatchEvent(new Event('visibilitychange')))
|
||||
expect(vi.getTimerCount()).toBe(0)
|
||||
|
||||
const frozen = status.textContent
|
||||
act(() => vi.advanceTimersByTime(800))
|
||||
expect(status.textContent).toBe(frozen)
|
||||
|
||||
Object.defineProperty(document, 'visibilityState', { configurable: true, value: 'visible' })
|
||||
act(() => document.dispatchEvent(new Event('visibilitychange')))
|
||||
expect(vi.getTimerCount()).toBe(1)
|
||||
|
||||
act(() => vi.advanceTimersByTime(80))
|
||||
expect(status.textContent).not.toBe(frozen)
|
||||
} finally {
|
||||
Object.defineProperty(document, 'visibilityState', { configurable: true, value: 'visible' })
|
||||
}
|
||||
expect(viewport()?.hasAttribute('data-paused')).toBe(false)
|
||||
})
|
||||
})
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
import { useEffect, useRef } from 'react'
|
||||
import './glyph-spinner.css'
|
||||
|
||||
import type { CSSProperties } from 'react'
|
||||
import spinners, { type BrailleSpinnerName as SpinnerName } from 'unicode-animations'
|
||||
|
||||
import { usePaneVisible } from '@/components/pane-shell/pane-visibility'
|
||||
import { createRendererLoopPauseController } from '@/lib/renderer-loop-pause'
|
||||
import { cn } from '@/lib/utils'
|
||||
|
||||
export type { SpinnerName }
|
||||
@@ -12,6 +13,18 @@ interface NormalisedSpinner {
|
||||
interval: number
|
||||
}
|
||||
|
||||
/**
|
||||
* The two custom properties `glyph-spinner.css` reads off the strip to size
|
||||
* `steps()` and the cycle duration. Naming them in a type, rather than casting
|
||||
* the inline style with a bare `as CSSProperties`, makes a typo in a property
|
||||
* name a compile error instead of a silently dead declaration that falls back
|
||||
* to the braille defaults at runtime.
|
||||
*/
|
||||
export type GlyphSpinnerVars = CSSProperties & {
|
||||
'--glyph-spinner-duration': string
|
||||
'--glyph-spinner-frames': number
|
||||
}
|
||||
|
||||
// Some spinners ship multi-character frames. Pull the first cell so each
|
||||
// frame fits in one monospace box — matches how the TUI uses them.
|
||||
const FRAMES_BY_NAME: Record<SpinnerName, NormalisedSpinner> = (() => {
|
||||
@@ -32,6 +45,10 @@ const FRAMES_BY_NAME: Record<SpinnerName, NormalisedSpinner> = (() => {
|
||||
interface GlyphSpinnerProps {
|
||||
ariaLabel?: string
|
||||
className?: string
|
||||
/** Freeze the animation while the spinner stays mounted — for a caller that
|
||||
* keeps it in the tree through a fade-out and does not want it animating
|
||||
* once it is no longer the thing being waited on. */
|
||||
paused?: boolean
|
||||
spinner?: SpinnerName
|
||||
}
|
||||
|
||||
@@ -41,69 +58,59 @@ interface GlyphSpinnerProps {
|
||||
* the desktop and terminal experiences read the same visually. Renders inside
|
||||
* an `inline-flex` cell with `leading-none` and `items-center` so it sits
|
||||
* vertically centred inside its parent's line-box.
|
||||
*
|
||||
* Animated entirely on the compositor — every frame is in the DOM from mount
|
||||
* and a transform keyframes animation scrolls between them. There is no timer
|
||||
* and no per-frame DOM write, because the ticker this replaced was scheduling
|
||||
* document-scale style recalculation on every tick (see glyph-spinner.css).
|
||||
*
|
||||
* The outer cell keeps the exact classes it always had, so consumer sizing
|
||||
* (`size-3`, `text-[0.75rem]`, colour, opacity) lands unchanged; the 1em-tall
|
||||
* clipping viewport is centred inside it by the same `items-center` that used
|
||||
* to centre the single glyph.
|
||||
*/
|
||||
export function GlyphSpinner({ ariaLabel = 'Loading', className, spinner = 'braille' }: GlyphSpinnerProps) {
|
||||
export function GlyphSpinner({
|
||||
ariaLabel = 'Loading',
|
||||
className,
|
||||
paused = false,
|
||||
spinner = 'braille'
|
||||
}: GlyphSpinnerProps) {
|
||||
const spin = FRAMES_BY_NAME[spinner] ?? FRAMES_BY_NAME.braille!
|
||||
const glyphRef = useRef<HTMLSpanElement>(null)
|
||||
// Pause when this surface is a hidden (kept-alive) tab: N mounted tabs each
|
||||
// ticking a setInterval burns CPU for pixels nobody can see.
|
||||
// animating burns CPU for pixels nobody can see. Window blur / minimize /
|
||||
// document-hidden are handled globally instead, by the
|
||||
// `:root[data-renderer-animations-paused]` rule in styles.css that
|
||||
// main.tsx's installRendererAnimationPauseState() drives — the same
|
||||
// mechanism every other continuous decorative animation already uses.
|
||||
// Note that only the primary window arms that attribute: a spinner mounted
|
||||
// in an overlay / quick / wake window keeps animating while its window is
|
||||
// blurred, exactly as the other decorative animations there do.
|
||||
const visible = usePaneVisible()
|
||||
|
||||
useEffect(() => {
|
||||
const glyph = glyphRef.current
|
||||
|
||||
if (!visible || !glyph) {
|
||||
return
|
||||
}
|
||||
|
||||
let frame = 0
|
||||
let timer: number | undefined
|
||||
let pauseController: ReturnType<typeof createRendererLoopPauseController> | undefined
|
||||
glyph.textContent = spin.frames[frame]
|
||||
|
||||
const stopAnimation = () => {
|
||||
if (timer === undefined) {
|
||||
return
|
||||
}
|
||||
|
||||
window.clearInterval(timer)
|
||||
timer = undefined
|
||||
}
|
||||
|
||||
const syncAnimation = () => {
|
||||
if (pauseController?.isPaused()) {
|
||||
stopAnimation()
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
if (timer !== undefined) {
|
||||
return
|
||||
}
|
||||
|
||||
timer = window.setInterval(() => {
|
||||
frame = (frame + 1) % spin.frames.length
|
||||
glyph.textContent = spin.frames[frame]
|
||||
}, spin.interval)
|
||||
}
|
||||
|
||||
pauseController = createRendererLoopPauseController(syncAnimation)
|
||||
syncAnimation()
|
||||
|
||||
return () => {
|
||||
pauseController.dispose()
|
||||
stopAnimation()
|
||||
}
|
||||
}, [spin, visible])
|
||||
const vars: GlyphSpinnerVars = {
|
||||
'--glyph-spinner-duration': `${spin.frames.length * spin.interval}ms`,
|
||||
'--glyph-spinner-frames': spin.frames.length
|
||||
}
|
||||
|
||||
return (
|
||||
<span
|
||||
aria-label={ariaLabel}
|
||||
className={cn('inline-flex items-center justify-center font-mono leading-none tabular-nums', className)}
|
||||
ref={glyphRef}
|
||||
role="status"
|
||||
>
|
||||
{spin.frames[0]}
|
||||
{/* Hidden from assistive tech: the accessible name is the label above.
|
||||
The frames are decorative, and this is a live region — the old
|
||||
implementation rewrote its text ~12x/second, which would announce a
|
||||
new glyph on every tick. */}
|
||||
<span aria-hidden="true" className="glyph-spinner" data-paused={paused || !visible ? 'true' : undefined}>
|
||||
<span className="glyph-spinner__strip" style={vars}>
|
||||
{spin.frames.map((frame, index) => (
|
||||
<span className="glyph-spinner__frame" key={`${index}:${frame}`}>
|
||||
{frame}
|
||||
</span>
|
||||
))}
|
||||
</span>
|
||||
</span>
|
||||
</span>
|
||||
)
|
||||
}
|
||||
|
||||
Vendored
+2
@@ -1311,6 +1311,8 @@ export interface HermesSelectPathsOptions {
|
||||
defaultPath?: string
|
||||
directories?: boolean
|
||||
multiple?: boolean
|
||||
/** Backend profile that produced defaultPath; Electron uses it for WSL gating. */
|
||||
profile?: string
|
||||
filters?: Array<{ name: string; extensions: string[] }>
|
||||
}
|
||||
|
||||
|
||||
@@ -76,7 +76,7 @@ describe('desktop filesystem facade', () => {
|
||||
})
|
||||
|
||||
it('uses local Electron filesystem methods in local mode', async () => {
|
||||
$connection.set({ mode: 'local' } as never)
|
||||
$connection.set({ mode: 'local', profile: 'team-local' } as never)
|
||||
|
||||
await expect(readDesktopDir('/work')).resolves.toEqual({
|
||||
entries: [{ name: 'local', path: '/local', isDirectory: true }]
|
||||
@@ -90,7 +90,7 @@ describe('desktop filesystem facade', () => {
|
||||
expect(readFileText).toHaveBeenCalledWith('/work/file.txt')
|
||||
expect(readFileDataUrl).toHaveBeenCalledWith('/work/file.txt')
|
||||
expect(gitRoot).toHaveBeenCalledWith('/work')
|
||||
expect(selectPaths).toHaveBeenCalledWith({ directories: true })
|
||||
expect(selectPaths).toHaveBeenCalledWith({ directories: true, profile: 'team-local' })
|
||||
expect(api).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
@@ -216,12 +216,12 @@ describe('desktop filesystem facade', () => {
|
||||
|
||||
it('uses the local Electron picker for remote file selection', async () => {
|
||||
const remoteSelect = vi.fn(async () => ['/remote/project'])
|
||||
$connection.set({ mode: 'remote' } as never)
|
||||
$connection.set({ mode: 'remote', profile: 'team-remote' } as never)
|
||||
setDesktopFsRemotePicker({ selectPaths: remoteSelect })
|
||||
|
||||
await expect(selectDesktopPaths({ directories: false, multiple: false })).resolves.toEqual(['/local'])
|
||||
|
||||
expect(selectPaths).toHaveBeenCalledWith({ directories: false, multiple: false })
|
||||
expect(selectPaths).toHaveBeenCalledWith({ directories: false, multiple: false, profile: 'team-remote' })
|
||||
expect(remoteSelect).not.toHaveBeenCalled()
|
||||
})
|
||||
|
||||
|
||||
@@ -201,13 +201,15 @@ export async function desktopFileDiff(repoRoot: string, filePath: string): Promi
|
||||
|
||||
export async function selectDesktopPaths(options?: HermesSelectPathsOptions): Promise<string[]> {
|
||||
const desktop = bridge()
|
||||
const profile = desktopFsProfile()
|
||||
const localOptions = profile ? { ...options, profile } : options
|
||||
|
||||
if (!isDesktopFsRemoteMode()) {
|
||||
return desktop.selectPaths(options)
|
||||
return desktop.selectPaths(localOptions)
|
||||
}
|
||||
|
||||
if (!options?.directories) {
|
||||
return desktop.selectPaths(options)
|
||||
return desktop.selectPaths(localOptions)
|
||||
}
|
||||
|
||||
return remotePicker ? remotePicker.selectPaths({ ...options, multiple: false }) : []
|
||||
|
||||
@@ -800,12 +800,25 @@ function durableGroupChatRooms(all = $groupChats.get()) {
|
||||
if (!room || !Array.isArray(room.log)) {
|
||||
continue
|
||||
}
|
||||
// Disband tombstones are runtime-only coordination state (they hold the
|
||||
// epoch bump for an in-flight drive). Persisting one would resurrect the
|
||||
// room as an empty record on the next load AND keep its name "taken" for
|
||||
// same-name recreates. Mirrors updateGroupChat's inline durable map.
|
||||
if (room.tombstone) {
|
||||
continue
|
||||
}
|
||||
durable[name] = {
|
||||
log: room.log,
|
||||
watermarks: room.watermarks || {},
|
||||
sessions: room.sessions || {},
|
||||
stranded: room.stranded || {},
|
||||
members: Array.isArray(room.members) ? room.members : [],
|
||||
// Immutable room identity: without this, a room merged in via the
|
||||
// remote-sync path (the only caller of this function) loses its
|
||||
// roomId on the next cold hydrate and falls back to legacy
|
||||
// name-keyed identity — same field updateGroupChat's inline map
|
||||
// already carries.
|
||||
roomId: typeof room.roomId === 'string' && room.roomId ? room.roomId : null,
|
||||
image: room.image || null,
|
||||
syncRevision: Math.max(0, Number(room.syncRevision || 0))
|
||||
}
|
||||
|
||||
@@ -196,7 +196,7 @@ function load(turnScript, { busyUntilResumeCall, clarifyUntilResumeCall, approva
|
||||
.replace(/^import .* from 'react\/jsx-runtime'\r?\n/m, '')
|
||||
.replace('export default {', 'globalThis.plugin = {')
|
||||
.concat(
|
||||
'\nglobalThis.__gc = { sendToGroupChat, runGroupChatRounds, harvestStrandedGroupReply, resolveGroupResponders, parseGroupChatMentions, rotateGroupSpeakers, isGroupPassText, formatGroupChatLine, buildGroupChatTurnPrompt, trimGroupChatLog, groupChatSyncSnapshot, groupChatGatewayJsonSize, mergeGroupChatSyncSnapshots, mergeRemoteGroupChatSnapshotIntoRooms, scheduleGroupChatServerSync, disbandGroupChat, renameGroupChat, updateGroupChat, ensureGroupChatSession, uniqueGroupChatName, liveGroupChatNames, openGroupChat, closeGroupChatMainTab, shouldRenderGroupChatInPane, syncGroupClarify, clearGroupClarify, answerGroupClarify, $groupClarify, $groupChats, $groupNeedsYou, $groupChatWorkspace, $groupMainTabsRev, $botMeta, GROUP_CHAT_MAX_ROUNDS, GROUP_CHAT_MAX_MESSAGES };\n'
|
||||
'\nglobalThis.__gc = { sendToGroupChat, runGroupChatRounds, harvestStrandedGroupReply, resolveGroupResponders, parseGroupChatMentions, rotateGroupSpeakers, isGroupPassText, formatGroupChatLine, buildGroupChatTurnPrompt, trimGroupChatLog, groupChatSyncSnapshot, groupChatGatewayJsonSize, mergeGroupChatSyncSnapshots, mergeRemoteGroupChatSnapshotIntoRooms, scheduleGroupChatServerSync, disbandGroupChat, renameGroupChat, updateGroupChat, durableGroupChatRooms, persistGroupChatRooms, ensureGroupChatSession, uniqueGroupChatName, liveGroupChatNames, openGroupChat, closeGroupChatMainTab, shouldRenderGroupChatInPane, syncGroupClarify, clearGroupClarify, answerGroupClarify, $groupClarify, $groupChats, $groupNeedsYou, $groupChatWorkspace, $groupMainTabsRev, $botMeta, GROUP_CHAT_MAX_ROUNDS, GROUP_CHAT_MAX_MESSAGES };\n'
|
||||
)
|
||||
vm.runInNewContext(source, context, { filename: 'plugin.js' })
|
||||
const storageWrites = new Map()
|
||||
@@ -1830,3 +1830,73 @@ test('source contract: approval card renders the command and routes approval.res
|
||||
assert.match(pluginSource, /kind: 'approval'/)
|
||||
assert.match(pluginSource, /wants to run a command/)
|
||||
})
|
||||
|
||||
test('durableGroupChatRooms excludes tombstoned rooms — the remote-merge persist path\'s own durable-map builder, independent of updateGroupChat\'s inline one', () => {
|
||||
const gc = load(() => '(pass)')
|
||||
|
||||
const durable = gc.durableGroupChatRooms({
|
||||
Live: { tombstone: true, log: [], watermarks: {}, epoch: 4, running: false },
|
||||
Keep: { log: [{ from: { kind: 'user' }, text: 'hi', at: 1 }], watermarks: {}, members: [] }
|
||||
})
|
||||
|
||||
assert.ok(!('Live' in durable), 'tombstone excluded from the remote-merge persist path too')
|
||||
assert.ok('Keep' in durable, 'real room still persisted')
|
||||
})
|
||||
|
||||
test('durableGroupChatRooms carries roomId — omitting it drops the durable id on the next cold hydrate', () => {
|
||||
const gc = load(() => '(pass)')
|
||||
|
||||
const durable = gc.durableGroupChatRooms({
|
||||
Team: {
|
||||
log: [{ from: { kind: 'user' }, text: 'hi', at: 1 }],
|
||||
watermarks: {},
|
||||
members: [],
|
||||
roomId: 'room-abc123'
|
||||
},
|
||||
Legacy: {
|
||||
log: [{ from: { kind: 'user' }, text: 'hi', at: 1 }],
|
||||
watermarks: {},
|
||||
members: []
|
||||
}
|
||||
})
|
||||
|
||||
assert.equal(durable.Team.roomId, 'room-abc123', 'immutable room identity must survive the remote-merge persist path')
|
||||
assert.equal(durable.Legacy.roomId, null, 'a room with no roomId persists an explicit null, not undefined')
|
||||
})
|
||||
|
||||
test('remote-merge reachability: a disband tombstone that survives a merge (gateway has not received the delete yet) is never written to storage', async () => {
|
||||
const gc = load(() => '(pass)')
|
||||
|
||||
// Simulate a drive still mid-turn at disband time, exactly like the
|
||||
// sibling disband test above — this room is a live tombstone in $groupChats.
|
||||
gc.$groupChats.set({
|
||||
Live: { tombstone: true, log: [], watermarks: {}, epoch: 4, running: false }
|
||||
})
|
||||
|
||||
// The remote gateway has NOT yet received the delete (plausible now that
|
||||
// sync fans out to every reachable default-profile gateway independently)
|
||||
// — its snapshot still carries a live copy of the room under the same
|
||||
// display name.
|
||||
const merged = gc.mergeRemoteGroupChatSnapshotIntoRooms(
|
||||
{
|
||||
rooms: {
|
||||
Live: {
|
||||
log: [{ from: { kind: 'member', name: 'research' }, text: 'still going', at: 1 }],
|
||||
members: [{ name: 'research' }]
|
||||
}
|
||||
}
|
||||
},
|
||||
gc.$groupChats.get()
|
||||
)
|
||||
|
||||
// The merge spreads `...existing` before its explicit field overrides,
|
||||
// none of which touch `tombstone` — so the flag survives into the merged
|
||||
// room. This is the reachability step: without it, durableGroupChatRooms
|
||||
// would never even see a tombstoned room from this path.
|
||||
assert.equal(merged.Live.tombstone, true, 'tombstone forwarded by the merge (reachability precondition)')
|
||||
|
||||
await gc.persistGroupChatRooms(merged)
|
||||
|
||||
const durable = gc.storageWrites.get('group-chats')
|
||||
assert.ok(durable && !('Live' in durable), 'merged tombstone must not resurrect as a persisted room')
|
||||
})
|
||||
|
||||
@@ -381,6 +381,21 @@ async function reconnectSecondary(entry: Secondary): Promise<void> {
|
||||
try {
|
||||
await openSecondary(entry)
|
||||
entry.reconnectAttempt = 0
|
||||
// The re-dialed backend may have respawned and re-minted runtime ids —
|
||||
// busy flags recorded from THIS socket's pre-drop events would then never
|
||||
// receive their terminal busy:false, leaving the session's running arc
|
||||
// armed forever (#53902/#73082 stale-flag half). Scoped: only runtimes
|
||||
// whose events arrived on this connection are reconciled; live work on
|
||||
// other sockets is untouched, and a genuinely live turn here re-asserts
|
||||
// busy on its next event. Lazy import: a static edge here closes a module
|
||||
// cycle (session-states → … → gateway) that leaves nanostores atoms
|
||||
// undefined at init for whichever module loads second. Best-effort catch:
|
||||
// under partial vi.mock('@/hermes') harnesses the transitive graph can
|
||||
// fail to load — a skipped reconcile there must not surface as an
|
||||
// unhandled rejection (the real graph always loads in production).
|
||||
void import('@/store/session-states')
|
||||
.then(({ reconcileBusyStatesOnReconnect }) => reconcileBusyStatesOnReconnect(entry.scope))
|
||||
.catch(() => undefined)
|
||||
} catch (error) {
|
||||
// The registry no longer knows this connection (removed while we were
|
||||
// backing off), or Electron's deletion guard reports the profile itself
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
import { registryBackendScopeKey } from '@hermes/shared'
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
|
||||
|
||||
import type { ClientSessionState } from '@/app/types'
|
||||
import { createClientSessionState } from '@/lib/chat-runtime'
|
||||
|
||||
import { $activeSessionId, $selectedStoredSessionId, $unreadFinishedSessionIds } from './session'
|
||||
import {
|
||||
$attentionSessionIds,
|
||||
$stalledSessionIds,
|
||||
$workingSessionIds,
|
||||
clearAllSessionStates,
|
||||
publishSessionState,
|
||||
reconcileBusyStatesOnReconnect,
|
||||
recordSessionEventScope,
|
||||
SESSION_WATCHDOG_TIMEOUT_MS
|
||||
} from './session-states'
|
||||
|
||||
function state(over: Partial<ClientSessionState> = {}): ClientSessionState {
|
||||
return { ...createClientSessionState(null), storedSessionId: 's1', ...over }
|
||||
}
|
||||
|
||||
// The stale-flag half of #53902/#73082: a backend respawn re-mints runtime
|
||||
// ids, so a pre-reconnect busy state never receives its terminal busy:false
|
||||
// and the session's running arc stays armed forever. The reconnect paths call
|
||||
// reconcileBusyStatesOnReconnect to retire those claims.
|
||||
describe('reconcileBusyStatesOnReconnect', () => {
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers()
|
||||
vi.setSystemTime(0)
|
||||
clearAllSessionStates()
|
||||
$unreadFinishedSessionIds.set([])
|
||||
$selectedStoredSessionId.set(null)
|
||||
$activeSessionId.set(null)
|
||||
})
|
||||
|
||||
afterEach(() => {
|
||||
vi.runOnlyPendingTimers()
|
||||
vi.useRealTimers()
|
||||
clearAllSessionStates()
|
||||
$unreadFinishedSessionIds.set([])
|
||||
$selectedStoredSessionId.set(null)
|
||||
$activeSessionId.set(null)
|
||||
})
|
||||
|
||||
it('clears a stale busy session on primary reconnect', () => {
|
||||
publishSessionState('rt1', state({ busy: true, storedSessionId: 's1' }))
|
||||
expect($workingSessionIds.get()).toContain('s1')
|
||||
|
||||
reconcileBusyStatesOnReconnect()
|
||||
|
||||
expect($workingSessionIds.get()).not.toContain('s1')
|
||||
})
|
||||
|
||||
it('disarms the stall watchdog with the busy claim', () => {
|
||||
publishSessionState('rt1', state({ busy: true, storedSessionId: 's1' }))
|
||||
|
||||
reconcileBusyStatesOnReconnect()
|
||||
|
||||
// Without reconcile the watchdog would fire and paint s1 stalled.
|
||||
vi.advanceTimersByTime(SESSION_WATCHDOG_TIMEOUT_MS + 1000)
|
||||
expect($stalledSessionIds.get()).not.toContain('s1')
|
||||
})
|
||||
|
||||
it('preserves needsInput — a blocking prompt is not a stale flag', () => {
|
||||
publishSessionState('rt1', state({ busy: true, needsInput: true, storedSessionId: 's1' }))
|
||||
expect($attentionSessionIds.get()).toContain('s1')
|
||||
|
||||
reconcileBusyStatesOnReconnect()
|
||||
|
||||
expect($workingSessionIds.get()).not.toContain('s1')
|
||||
expect($attentionSessionIds.get()).toContain('s1')
|
||||
})
|
||||
|
||||
it('primary reconcile leaves registry-scoped sessions alone', () => {
|
||||
const scope = registryBackendScopeKey('connA', 'default')
|
||||
publishSessionState('rtA', state({ busy: true, storedSessionId: 'sA' }))
|
||||
recordSessionEventScope({ connectionId: 'connA', profile: 'default', session_id: 'rtA' })
|
||||
publishSessionState('rtLocal', state({ busy: true, storedSessionId: 'sLocal' }))
|
||||
|
||||
reconcileBusyStatesOnReconnect()
|
||||
|
||||
expect($workingSessionIds.get()).toContain('sA')
|
||||
expect($workingSessionIds.get()).not.toContain('sLocal')
|
||||
|
||||
// And the scoped variant clears ONLY its own connection's sessions.
|
||||
reconcileBusyStatesOnReconnect(scope)
|
||||
expect($workingSessionIds.get()).not.toContain('sA')
|
||||
})
|
||||
|
||||
it('scoped reconcile does not touch other connections or the primary', () => {
|
||||
publishSessionState('rtA', state({ busy: true, storedSessionId: 'sA' }))
|
||||
recordSessionEventScope({ connectionId: 'connA', profile: 'default', session_id: 'rtA' })
|
||||
publishSessionState('rtB', state({ busy: true, storedSessionId: 'sB' }))
|
||||
recordSessionEventScope({ connectionId: 'connB', profile: 'default', session_id: 'rtB' })
|
||||
publishSessionState('rtLocal', state({ busy: true, storedSessionId: 'sLocal' }))
|
||||
|
||||
reconcileBusyStatesOnReconnect(registryBackendScopeKey('connA', 'default'))
|
||||
|
||||
expect($workingSessionIds.get()).not.toContain('sA')
|
||||
expect($workingSessionIds.get()).toContain('sB')
|
||||
expect($workingSessionIds.get()).toContain('sLocal')
|
||||
})
|
||||
|
||||
it('a live turn re-asserting busy after reconcile re-arms the arc', () => {
|
||||
const s = state({ busy: true, storedSessionId: 's1' })
|
||||
publishSessionState('rt1', s)
|
||||
reconcileBusyStatesOnReconnect()
|
||||
expect($workingSessionIds.get()).not.toContain('s1')
|
||||
|
||||
// The still-alive backend's next event republishes busy under a live id.
|
||||
publishSessionState('rt2', state({ busy: true, storedSessionId: 's1' }))
|
||||
|
||||
expect($workingSessionIds.get()).toContain('s1')
|
||||
})
|
||||
})
|
||||
@@ -373,6 +373,49 @@ export function clearAllSessionStates() {
|
||||
$sessionStates.set({})
|
||||
}
|
||||
|
||||
/** Downgrade cached busy/awaiting states after a gateway reconnect.
|
||||
*
|
||||
* A respawned backend re-mints runtime ids (the same fact that drives
|
||||
* resetTileRuntimeBindings), so a pre-reconnect `busy` can never receive its
|
||||
* terminal `busy: false` publish — the runtime id it would arrive under is
|
||||
* dead. Left alone, that state keeps its session in $workingSessionIds
|
||||
* forever: the sidebar running arc and agents-panel "running" chrome lie for
|
||||
* hours after the turn actually ended (#53902, #73082 — stale-flag half).
|
||||
*
|
||||
* `scope` picks which socket's sessions to reconcile, keyed by the event-
|
||||
* source scope recorded at fan-in: a SECONDARY (registry) reconnect passes
|
||||
* its composite scope and touches only runtimes that arrived on that socket;
|
||||
* the PRIMARY reconnect passes undefined and touches only scope-less
|
||||
* runtimes (primary/local events record no scope). Neither can clear live
|
||||
* work riding a different, still-healthy connection.
|
||||
*
|
||||
* Direction of failure is deliberate: a turn that IS still live (transient
|
||||
* socket blip, same backend) re-asserts busy on its next event or inflight
|
||||
* snapshot within a beat, so at worst its arc blinks once. A dead turn's
|
||||
* state, by contrast, would never clear on its own. `needsInput` is left
|
||||
* untouched — a blocking prompt is the one claim the user must explicitly
|
||||
* answer, and post-reconnect refresh re-asserts or retires it via its own
|
||||
* path. Transition side-effects run through publishSessionState, so
|
||||
* watchdogs disarm, stall hints drop, and settle/unread bookkeeping stays
|
||||
* consistent. */
|
||||
export function reconcileBusyStatesOnReconnect(scope?: string) {
|
||||
const states = $sessionStates.get()
|
||||
|
||||
for (const [runtimeId, state] of Object.entries(states)) {
|
||||
if (!state || (!state.busy && !state.awaitingResponse)) {
|
||||
continue
|
||||
}
|
||||
|
||||
const recorded = sessionScopeByRuntimeId.get(runtimeId)
|
||||
|
||||
if (scope === undefined ? recorded !== undefined : recorded !== scope) {
|
||||
continue
|
||||
}
|
||||
|
||||
publishSessionState(runtimeId, { ...state, awaitingResponse: false, busy: false })
|
||||
}
|
||||
}
|
||||
|
||||
// Derived per-session status sets — pure projections of `$sessionStates` (which
|
||||
// holds `busy`/`needsInput` per runtime), keeping the data flow one-directional:
|
||||
// gateway event → cache → $sessionStates → computed views.
|
||||
|
||||
@@ -32,7 +32,17 @@
|
||||
behind another app. main.tsx owns this attribute only for the primary
|
||||
window; wake/pet overlays retain their purpose-built visibility behavior. */
|
||||
:root[data-renderer-animations-paused]
|
||||
:is(.shimmer, .quest-glow, .pet-egg, .pet-egg__glow, .pet-egg-shadow, .pet-wobble, .progress-slide, .kanban-arc),
|
||||
:is(
|
||||
.shimmer,
|
||||
.quest-glow,
|
||||
.pet-egg,
|
||||
.pet-egg__glow,
|
||||
.pet-egg-shadow,
|
||||
.pet-wobble,
|
||||
.progress-slide,
|
||||
.kanban-arc,
|
||||
.glyph-spinner__strip
|
||||
),
|
||||
:root[data-renderer-animations-paused] .arc-border::before,
|
||||
:root[data-renderer-animations-paused]
|
||||
[data-slot='aui_assistant-message-content']
|
||||
@@ -982,11 +992,19 @@
|
||||
}
|
||||
|
||||
@keyframes arc-border {
|
||||
/* Compositor-only travel. The ::before layer is 300% × 300% of the host
|
||||
(mirroring the old `background-size: 300%`), so translating it from
|
||||
-10% → -50% of its own size reproduces the old
|
||||
`background-position: 15% → 75%` exactly (offset = -2·W·p), while
|
||||
`transform` animates on the compositor with zero per-frame style
|
||||
recalc/paint. The old background-position version cost ~3,600
|
||||
main-thread style recalcs per minute per arc, pinning idle renderers
|
||||
(#53902, #73082). */
|
||||
0% {
|
||||
background-position: 15% 15%;
|
||||
transform: translate(-10%, -10%);
|
||||
}
|
||||
100% {
|
||||
background-position: 75% 75%;
|
||||
transform: translate(-50%, -50%);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1067,8 +1085,14 @@
|
||||
.arc-border::before {
|
||||
content: '';
|
||||
position: absolute;
|
||||
inset: 0;
|
||||
border-radius: inherit;
|
||||
top: 0;
|
||||
left: 0;
|
||||
/* The gradient layer IS 300% of the ring (the old version kept a host-sized
|
||||
::before and slid a 300%-sized background through it). Sizing the layer
|
||||
itself lets the travel be a `transform` — composited, no main-thread
|
||||
style/paint work — while the host's mask + overflow clip it to the ring. */
|
||||
width: 300%;
|
||||
height: 300%;
|
||||
background: linear-gradient(
|
||||
var(--arc-angle),
|
||||
transparent 0%,
|
||||
@@ -1085,7 +1109,7 @@
|
||||
var(--arc-c0) 95%,
|
||||
var(--arc-c1) 100%
|
||||
);
|
||||
background-size: 300% 300%;
|
||||
will-change: transform;
|
||||
animation: arc-border var(--arc-duration) linear infinite;
|
||||
}
|
||||
|
||||
@@ -2489,15 +2513,21 @@ button[data-slot='aui_msg-reactions'] svg {
|
||||
flow). Color/positioning come from the primitive; this owns the animation. */
|
||||
.progress-slide {
|
||||
width: 40%;
|
||||
/* left is pinned; travel is compositor-only (see arc-border note). */
|
||||
left: 0;
|
||||
will-change: transform;
|
||||
animation: progress-slide 1.15s ease-in-out infinite;
|
||||
}
|
||||
|
||||
@keyframes progress-slide {
|
||||
/* Old version animated `left: -42% → 100%`, forcing main-thread layout
|
||||
every frame for the whole hatch flow. The block is 40% of its track, so
|
||||
the same travel in units of its own width is -105% → 250%. */
|
||||
0% {
|
||||
left: -42%;
|
||||
transform: translateX(-105%);
|
||||
}
|
||||
100% {
|
||||
left: 100%;
|
||||
transform: translateX(250%);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
paultaki
|
||||
@@ -0,0 +1 @@
|
||||
qixuancao
|
||||
@@ -0,0 +1 @@
|
||||
vinsew
|
||||
+9
-6
@@ -358,11 +358,9 @@ class CronPromptInjectionBlocked(Exception):
|
||||
def _resolve_cron_disabled_toolsets(cfg: dict) -> list[str]:
|
||||
"""Toolsets a cron-spawned agent must never receive.
|
||||
|
||||
Three toolsets are always disabled in cron context regardless of config:
|
||||
Two toolsets are always disabled in cron context regardless of config:
|
||||
- ``messaging`` — interactive, needs a live gateway session
|
||||
- ``clarify`` — interactive, blocks waiting for user input
|
||||
- ``memory`` — cron agents are constructed with ``skip_memory=True``, so
|
||||
exposing this tool only gives the model an unbacked tool that fails
|
||||
|
||||
``cronjob`` is policy-denied by default (loop prevention, not a security
|
||||
boundary) and config-gated: setting ``cron.allow_agent_scheduling: true``
|
||||
@@ -377,9 +375,9 @@ def _resolve_cron_disabled_toolsets(cfg: dict) -> list[str]:
|
||||
"""
|
||||
cron_cfg = (cfg or {}).get("cron") or {}
|
||||
if cron_cfg.get("allow_agent_scheduling"):
|
||||
disabled = ["messaging", "clarify", "memory"]
|
||||
disabled = ["messaging", "clarify"]
|
||||
else:
|
||||
disabled = ["cronjob", "messaging", "clarify", "memory"]
|
||||
disabled = ["cronjob", "messaging", "clarify"]
|
||||
agent_cfg = (cfg or {}).get("agent") or {}
|
||||
from agent.skill_utils import parse_config_string_list
|
||||
|
||||
@@ -496,6 +494,7 @@ def _resolve_job_reasoning_config(job: dict, cfg: dict, model: str) -> dict | No
|
||||
)
|
||||
return resolve_reasoning_config(cfg if isinstance(cfg, dict) else {}, str(model))
|
||||
|
||||
|
||||
# Valid delivery platforms — used to validate user-supplied platform names
|
||||
# in cron delivery targets, preventing env var enumeration via crafted names.
|
||||
_KNOWN_DELIVERY_PLATFORMS = frozenset({
|
||||
@@ -5813,7 +5812,11 @@ def run_job(
|
||||
# Without a workdir, keep cwd context discovery disabled.
|
||||
skip_context_files=not bool(_job_workdir),
|
||||
load_soul_identity=True,
|
||||
skip_memory=True, # Cron system prompts would corrupt user representations
|
||||
# Memory is enabled for cron agents like any other agent run:
|
||||
# MEMORY.md / USER.md load into the system prompt and the memory
|
||||
# tool follows normal toolset resolution, so jobs benefit from
|
||||
# (and can update) the user's persistent memory.
|
||||
skip_memory=False,
|
||||
skip_background_review=True, # Cron has no human-in-the-loop need for skill/memory review forks (~30K tok/event)
|
||||
platform="cron",
|
||||
session_id=_cron_session_id,
|
||||
|
||||
@@ -14,8 +14,8 @@ Provides subcommands for:
|
||||
import os
|
||||
import sys
|
||||
|
||||
__version__ = "0.20.4"
|
||||
__release_date__ = "2026.8.18"
|
||||
__version__ = "0.20.5"
|
||||
__release_date__ = "2026.8.19"
|
||||
|
||||
|
||||
def _ensure_utf8():
|
||||
|
||||
@@ -7096,6 +7096,25 @@ def get_api_key_provider_status(provider_id: str) -> Dict[str, Any]:
|
||||
if not pconfig or pconfig.auth_type != "api_key":
|
||||
return {"configured": False}
|
||||
|
||||
# Keyless providers (opencode-free) are served anonymously: no credential
|
||||
# exists, so every install counts as configured/logged in. Derived from
|
||||
# the HermesOverlay keyless flag — the same source the provider catalog
|
||||
# and GUI contract tests use.
|
||||
try:
|
||||
from hermes_cli.providers import HERMES_OVERLAYS
|
||||
_overlay = HERMES_OVERLAYS.get(provider_id)
|
||||
except Exception:
|
||||
_overlay = None
|
||||
if _overlay is not None and getattr(_overlay, "keyless", False):
|
||||
return {
|
||||
"configured": True,
|
||||
"provider": provider_id,
|
||||
"name": pconfig.name,
|
||||
"key_source": "keyless",
|
||||
"base_url": pconfig.inference_base_url,
|
||||
"logged_in": True,
|
||||
}
|
||||
|
||||
api_key = ""
|
||||
key_source = ""
|
||||
api_key, key_source = _resolve_api_key_provider_secret(provider_id, pconfig)
|
||||
|
||||
+216
-76
@@ -107,11 +107,14 @@ def _get_service_pids(all_profiles: bool = False) -> set:
|
||||
returns (true for both systemd and launchd in practice).
|
||||
|
||||
``all_profiles`` widens the launchd branch to every installed
|
||||
``ai.hermes.gateway*`` agent — the update path needs the whole fleet
|
||||
excluded from its sweep so sibling-profile launchd gateways found by the
|
||||
ps scan aren't misclassified as manual processes (#73626). Default-scope
|
||||
callers (``gateway status``, cron checks) keep seeing only the current
|
||||
profile's service.
|
||||
``ai.hermes.gateway*`` LaunchAgent — the update path needs the whole
|
||||
fleet excluded from its sweep (#41403, #73626): sibling-profile launchd
|
||||
gateways found by the (BSD-fixed) ps scan must not be misclassified as
|
||||
manual processes and killed. Default-scope callers (``gateway status``,
|
||||
cron checks) keep seeing only the current profile's service; the orphan
|
||||
reaper passes all_profiles=True for the same friendly-fire reason. The
|
||||
systemd branch has always been fleet-wide (``hermes-gateway*``) and is
|
||||
unaffected.
|
||||
"""
|
||||
pids: set = set()
|
||||
|
||||
@@ -155,13 +158,29 @@ def _get_service_pids(all_profiles: bool = False) -> set:
|
||||
|
||||
# --- launchd (macOS) ---
|
||||
if is_macos():
|
||||
try:
|
||||
if all_profiles:
|
||||
# Enumerate every ai.hermes.gateway* agent across profiles
|
||||
# so the update sweep's exclude set is complete (#73626).
|
||||
# Without this, sibling-profile launchd gateways found by the
|
||||
# (now-working) ps scan would be misclassified as manual and
|
||||
# killed, racing with KeepAlive → duplicate gateways.
|
||||
labels = {get_launchd_label()}
|
||||
if all_profiles:
|
||||
# Every gateway LaunchAgent, not just the invoking profile's —
|
||||
# mirrors the systemd branch's ``hermes-gateway*`` pattern above.
|
||||
# The update path restarts the whole fleet, and its stale-process
|
||||
# sweep must not mistake a sibling service's fresh PID for a
|
||||
# manual gateway it should kill (#41403).
|
||||
labels.update(launchd_gateway_labels_for_install())
|
||||
for label in sorted(labels):
|
||||
try:
|
||||
_domain, pid = _locate_launchd_gateway_service(label)
|
||||
except subprocess.TimeoutExpired:
|
||||
continue
|
||||
if pid is not None and pid > 0:
|
||||
pids.add(pid)
|
||||
if all_profiles:
|
||||
# Belt-and-suspenders for the EXCLUDE use case (#74075): a bare
|
||||
# ``launchctl list`` prefix scan also catches ai.hermes.gateway*
|
||||
# agents the label derivation can't map (renamed profiles, other
|
||||
# installs sharing this user). Over-inclusion is safe here —
|
||||
# these PIDs are only ever protected from the kill sweep, never
|
||||
# targeted. Restart paths use the label-derived set only.
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["launchctl", "list"],
|
||||
capture_output=True,
|
||||
@@ -180,33 +199,8 @@ def _get_service_pids(all_profiles: bool = False) -> set:
|
||||
pids.add(pid)
|
||||
except ValueError:
|
||||
pass
|
||||
else:
|
||||
label = get_launchd_label()
|
||||
result = subprocess.run(
|
||||
["launchctl", "list", label],
|
||||
capture_output=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=5,
|
||||
)
|
||||
if result.returncode == 0:
|
||||
# Try plist format first (macOS 26+): "PID" = <N>;
|
||||
pid = _parse_launchd_pid_from_list_output(result.stdout)
|
||||
if pid is not None and pid > 0:
|
||||
pids.add(pid)
|
||||
else:
|
||||
# Fall back to legacy tab-separated format:
|
||||
# "PID\tStatus\tLabel"
|
||||
for line in result.stdout.strip().splitlines():
|
||||
parts = line.split()
|
||||
if len(parts) >= 3 and parts[2] == label:
|
||||
try:
|
||||
pid = int(parts[0])
|
||||
if pid > 0:
|
||||
pids.add(pid)
|
||||
except ValueError:
|
||||
pass
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
pass
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
pass
|
||||
|
||||
return pids
|
||||
|
||||
@@ -1437,6 +1431,87 @@ def _parse_launchd_pid_from_list_output(output: str) -> int | None:
|
||||
return None
|
||||
|
||||
|
||||
def _parse_launchd_pid_from_print_output(output: str) -> int | None:
|
||||
"""Extract the live PID from ``launchctl print`` output (``pid = <N>``).
|
||||
|
||||
A bootstrapped-but-not-running service prints no ``pid =`` line; the
|
||||
first (service-level) occurrence wins over any nested endpoint state.
|
||||
Returns ``None`` when no PID is found or the PID is non-positive.
|
||||
"""
|
||||
for line in output.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("pid = "):
|
||||
try:
|
||||
pid = int(stripped[len("pid = "):].strip())
|
||||
return pid if pid > 0 else None
|
||||
except ValueError:
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _launchd_print_service_pid(domain: str, label: str) -> tuple[bool, int | None]:
|
||||
"""Return ``(loaded, pid)`` for ``domain/label`` via ``launchctl print``.
|
||||
|
||||
Domain-explicit on purpose: legacy ``launchctl list`` infers its domain
|
||||
from the caller's execution context, which is exactly the ambiguity that
|
||||
sank the first fleet-restart attempt (#41403 review). ``TimeoutExpired``
|
||||
propagates — fleet-restart callers own per-label failure accounting (a
|
||||
wedged launchctl call must be reported, not read as "unloaded").
|
||||
"""
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["launchctl", "print", f"{domain}/{label}"],
|
||||
capture_output=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=5,
|
||||
)
|
||||
except FileNotFoundError:
|
||||
return (False, None)
|
||||
if result.returncode != 0:
|
||||
return (False, None)
|
||||
return (True, _parse_launchd_pid_from_print_output(result.stdout))
|
||||
|
||||
|
||||
def _launchd_service_registered(label: str) -> bool:
|
||||
"""True when launchd knows ``label`` (``launchctl list <label>`` exit 0).
|
||||
|
||||
Registration is domain-agnostic and — unlike the ``launchctl print``
|
||||
domain probes in ``_locate_launchd_gateway_service`` — stays true on
|
||||
macOS 26+ hosts whose per-user domains reject service management, so
|
||||
the update path can still hand the label to ``launchd_restart()``,
|
||||
which owns that fallback. ``FileNotFoundError``/``TimeoutExpired``
|
||||
propagate: the caller treats gate errors as a best-effort skip,
|
||||
matching the pre-fleet inline behavior.
|
||||
"""
|
||||
result = subprocess.run(
|
||||
["launchctl", "list", label],
|
||||
capture_output=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=5,
|
||||
)
|
||||
return result.returncode == 0
|
||||
|
||||
|
||||
def _locate_launchd_gateway_service(label: str) -> tuple[str | None, int | None]:
|
||||
"""Return ``(domain, pid)`` for ``label``, probing both per-user domains.
|
||||
|
||||
Probes ``gui/<uid>`` first (Aqua sessions), then ``user/<uid>``
|
||||
(Background/SSH sessions). ``domain`` is None when the label is not
|
||||
bootstrapped in either; ``pid`` is None when the service has no live
|
||||
process. Sibling profile services resolve independently — a fleet can
|
||||
legitimately mix domains (a profile installed over SSH lands in
|
||||
``user/<uid>`` while the rest live in ``gui/<uid>``), so the current
|
||||
profile's cached domain (``_launchd_domain()``) is never consulted.
|
||||
``TimeoutExpired`` propagates (see ``_launchd_print_service_pid``).
|
||||
"""
|
||||
uid = os.getuid() # windows-footgun: ok — POSIX launchd (macOS) helper, never invoked on Windows
|
||||
for domain in (f"gui/{uid}", f"user/{uid}"):
|
||||
loaded, pid = _launchd_print_service_pid(domain, label)
|
||||
if loaded:
|
||||
return (domain, pid)
|
||||
return (None, None)
|
||||
|
||||
|
||||
def _probe_launchd_service_running() -> bool:
|
||||
"""Return True when launchd is actively supervising the gateway process.
|
||||
|
||||
@@ -3018,6 +3093,36 @@ def get_launchd_plist_path() -> Path:
|
||||
return _launchd_user_home() / "Library" / "LaunchAgents" / f"{name}.plist"
|
||||
|
||||
|
||||
def launchd_gateway_labels_for_install() -> list[str]:
|
||||
"""Return the launchd gateway label for every profile of THIS install.
|
||||
|
||||
Derived from the install's profile layout (rooted at
|
||||
``get_default_hermes_root()``), NOT by globbing ``~/Library/LaunchAgents``:
|
||||
the LaunchAgents directory is shared per-user, so a sandboxed
|
||||
``HERMES_HOME`` (tests, capture sandboxes, side-by-side installs) must
|
||||
never enumerate — let alone restart — another install's fleet.
|
||||
|
||||
Root label first, then profile labels sorted by name. Profile names
|
||||
that cannot map to a service suffix (see ``_profile_suffix``'s naming
|
||||
rule) are skipped — ``gateway install`` could never have created a
|
||||
predictable label for them. Profiles without an installed gateway are
|
||||
harmless to include: their labels simply aren't bootstrapped and
|
||||
callers skip them after a failed locate.
|
||||
"""
|
||||
import re as _re
|
||||
|
||||
from hermes_cli.profiles import list_profiles
|
||||
|
||||
root_label: list[str] = []
|
||||
profile_labels: list[str] = []
|
||||
for profile in list_profiles():
|
||||
if profile.is_default:
|
||||
root_label.append("ai.hermes.gateway")
|
||||
elif _re.match(r"^[a-z0-9][a-z0-9_-]{0,63}$", profile.name):
|
||||
profile_labels.append(f"ai.hermes.gateway-{profile.name}")
|
||||
return root_label + sorted(profile_labels)
|
||||
|
||||
|
||||
def _detect_venv_dir() -> Path | None:
|
||||
"""Detect the active virtualenv directory.
|
||||
|
||||
@@ -4231,52 +4336,35 @@ def get_launchd_label() -> str:
|
||||
_resolved_launchd_domain: str | None = None
|
||||
|
||||
|
||||
def _launchd_domain() -> str:
|
||||
"""Return the launchd domain that actually manages the gateway service.
|
||||
def _probe_launchd_domain_for_label(label: str) -> str:
|
||||
"""Resolve the launchd domain that manages ``label`` — uncached, per label.
|
||||
|
||||
Probes ``gui/<uid>`` first (Aqua sessions), then ``user/<uid>``
|
||||
(Background/SSH sessions). When neither domain contains a loaded
|
||||
service, falls back to ``launchctl managername`` as a heuristic.
|
||||
|
||||
The result is cached for the lifetime of the process so that repeated
|
||||
calls (``start``, ``stop``, ``restart``) use a consistent domain.
|
||||
|
||||
See #40831, #23387.
|
||||
Sibling profile services resolve independently: a fleet can legitimately
|
||||
mix domains (a profile installed over SSH lands in ``user/<uid>`` while
|
||||
the rest live in ``gui/<uid>``), so the current profile's cached domain
|
||||
(``_launchd_domain()``) must never be reused for another label.
|
||||
"""
|
||||
global _resolved_launchd_domain
|
||||
if _resolved_launchd_domain is not None:
|
||||
return _resolved_launchd_domain
|
||||
|
||||
uid = os.getuid() # windows-footgun: ok — POSIX launchd (macOS) helper, never invoked on Windows
|
||||
label = get_launchd_label()
|
||||
gui_domain = f"gui/{uid}"
|
||||
user_domain = f"user/{uid}"
|
||||
|
||||
# 1. Probe gui/<uid> first — in Aqua sessions the service is loaded here.
|
||||
try:
|
||||
subprocess.run(
|
||||
["launchctl", "print", f"{gui_domain}/{label}"],
|
||||
check=True,
|
||||
timeout=5,
|
||||
capture_output=True,
|
||||
)
|
||||
_resolved_launchd_domain = gui_domain
|
||||
return gui_domain
|
||||
except (subprocess.CalledProcessError, subprocess.TimeoutExpired, FileNotFoundError):
|
||||
pass
|
||||
|
||||
# 2. Probe user/<uid> — in Background/SSH sessions this is the working domain.
|
||||
try:
|
||||
subprocess.run(
|
||||
["launchctl", "print", f"{user_domain}/{label}"],
|
||||
check=True,
|
||||
timeout=5,
|
||||
capture_output=True,
|
||||
)
|
||||
_resolved_launchd_domain = user_domain
|
||||
return user_domain
|
||||
except (subprocess.CalledProcessError, subprocess.TimeoutExpired, FileNotFoundError):
|
||||
pass
|
||||
# 2. Then user/<uid> — in Background/SSH sessions this is the working domain.
|
||||
for domain in (gui_domain, user_domain):
|
||||
try:
|
||||
subprocess.run(
|
||||
["launchctl", "print", f"{domain}/{label}"],
|
||||
check=True,
|
||||
timeout=5,
|
||||
capture_output=True,
|
||||
)
|
||||
return domain
|
||||
except (subprocess.CalledProcessError, subprocess.TimeoutExpired, FileNotFoundError):
|
||||
pass
|
||||
|
||||
# 3. Neither domain has the service loaded — use managername as heuristic.
|
||||
# Aqua → gui/<uid>, anything else (Background, loginwindow) → user/<uid>.
|
||||
@@ -4288,17 +4376,32 @@ def _launchd_domain() -> str:
|
||||
timeout=5,
|
||||
)
|
||||
if "Aqua" in (result.stdout or ""):
|
||||
_resolved_launchd_domain = gui_domain
|
||||
return gui_domain
|
||||
except (subprocess.CalledProcessError, subprocess.TimeoutExpired, FileNotFoundError):
|
||||
pass
|
||||
|
||||
# 4. Default to user/<uid> (matches the pre-probing behavior for
|
||||
# Background/SSH sessions and is the recommended domain on macOS 26+).
|
||||
_resolved_launchd_domain = user_domain
|
||||
return user_domain
|
||||
|
||||
|
||||
def _launchd_domain() -> str:
|
||||
"""Return the launchd domain that actually manages the gateway service.
|
||||
|
||||
Per-label probing lives in ``_probe_launchd_domain_for_label``; this
|
||||
wrapper resolves the *current* profile's label and caches the result for
|
||||
the lifetime of the process so that repeated calls (``start``, ``stop``,
|
||||
``restart``) use a consistent domain.
|
||||
|
||||
See #40831, #23387.
|
||||
"""
|
||||
global _resolved_launchd_domain
|
||||
if _resolved_launchd_domain is not None:
|
||||
return _resolved_launchd_domain
|
||||
_resolved_launchd_domain = _probe_launchd_domain_for_label(get_launchd_label())
|
||||
return _resolved_launchd_domain
|
||||
|
||||
|
||||
# On macOS, exit code 125 ("Domain does not support specified action") and
|
||||
# 3/113 ("Could not find service") all mean the job isn't currently loaded in
|
||||
# the target domain, so start/restart should re-bootstrap the plist and retry.
|
||||
@@ -5163,6 +5266,43 @@ def _wait_for_gateway_exit(
|
||||
return True
|
||||
|
||||
|
||||
def _launchd_kickstart(label: str, domain: str) -> None:
|
||||
"""Hard-restart ``domain/label`` via ``launchctl kickstart -k``.
|
||||
|
||||
Raises ``CalledProcessError``/``TimeoutExpired`` — callers own the
|
||||
per-label failure accounting during fleet restarts.
|
||||
"""
|
||||
subprocess.run(
|
||||
["launchctl", "kickstart", "-k", f"{domain}/{label}"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True, encoding='utf-8', errors='replace',
|
||||
timeout=90,
|
||||
)
|
||||
|
||||
|
||||
def _wait_for_launchd_service_pid(
|
||||
label: str, old_pid: int | None, timeout: float = 10.0, *, domain: str
|
||||
) -> bool:
|
||||
"""Poll ``domain/label`` until the service runs on a fresh PID.
|
||||
|
||||
launchd's exit → ``KeepAlive`` respawn transition is not instantaneous;
|
||||
a one-shot check races that window and falsely reports the service as
|
||||
down (same rationale as the systemd ``is-active`` poll in the update
|
||||
path). Poll every 0.5s up to ``timeout`` seconds before giving up.
|
||||
``TimeoutExpired`` from launchctl propagates — callers own per-label
|
||||
failure accounting.
|
||||
"""
|
||||
deadline = time.monotonic() + max(timeout, 0.5)
|
||||
while True:
|
||||
_loaded, pid = _launchd_print_service_pid(domain, label)
|
||||
if pid is not None and pid > 0 and pid != old_pid:
|
||||
return True
|
||||
if time.monotonic() >= deadline:
|
||||
return False
|
||||
time.sleep(0.5)
|
||||
|
||||
|
||||
def launchd_restart():
|
||||
label = get_launchd_label()
|
||||
target = f"{_launchd_domain()}/{label}"
|
||||
|
||||
@@ -691,11 +691,27 @@ def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list
|
||||
if _raw_config_has_enabled_moa_preset():
|
||||
kept.append(row)
|
||||
continue
|
||||
if _provider_is_keyless(slug):
|
||||
# Keyless providers (opencode-free) require no configuration at
|
||||
# all — there is nothing to "explicitly configure", and hiding
|
||||
# them would defeat their purpose (zero-setup discoverability).
|
||||
kept.append(row)
|
||||
continue
|
||||
if is_provider_explicitly_configured(slug):
|
||||
kept.append(row)
|
||||
return kept
|
||||
|
||||
|
||||
def _provider_is_keyless(slug: str) -> bool:
|
||||
"""True when the provider's Hermes overlay declares it keyless."""
|
||||
try:
|
||||
from hermes_cli.providers import HERMES_OVERLAYS
|
||||
overlay = HERMES_OVERLAYS.get(slug)
|
||||
return bool(overlay is not None and getattr(overlay, "keyless", False))
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _raw_config_has_enabled_moa_preset() -> bool:
|
||||
"""Return True when the user's raw config explicitly enables MoA.
|
||||
|
||||
|
||||
@@ -10079,6 +10079,22 @@ def cmd_update(args):
|
||||
managed_error("update Hermes Agent")
|
||||
return
|
||||
|
||||
# --plan is read-only and deployment-kind aware, so it runs BEFORE the
|
||||
# docker/nix/apt refusal gates: on an image-managed or package-managed
|
||||
# install the plan itself reports "not updatable in place" plus the
|
||||
# right mechanism — strictly more useful than the bare refusal text.
|
||||
if getattr(args, "plan", False):
|
||||
# Read-only plan phase (#91277 Phase 2): inventory every running
|
||||
# Hermes runtime across profiles, its supervisor, and its running
|
||||
# code version — without mutating anything. Safe on a live fleet.
|
||||
from hermes_cli.update_inventory import (
|
||||
collect_runtime_inventory,
|
||||
print_update_plan,
|
||||
)
|
||||
|
||||
print_update_plan(collect_runtime_inventory())
|
||||
return
|
||||
|
||||
# Docker users can't ``git pull`` — the image excludes ``.git`` from
|
||||
# the build context. Bail with a friendly explanation pointing at
|
||||
# ``docker pull`` BEFORE any of the apply-path / check-path branches
|
||||
@@ -10130,6 +10146,38 @@ def cmd_update(args):
|
||||
|
||||
try:
|
||||
_self()._cmd_update_impl(args, gateway_mode=gateway_mode)
|
||||
except SystemExit as _update_exit:
|
||||
# Receipt boundary (#91283 review): the impl has many early
|
||||
# sys.exit paths (concurrent-instance preflight, venv-holder
|
||||
# refusal, head-pinned no-op, fetch failure) that never reach an
|
||||
# inner finalize. Persist any still-open receipt with the real
|
||||
# exit code, then let the exit proceed unchanged. No-op when an
|
||||
# inner path already finalized (exactly-once by construction).
|
||||
try:
|
||||
from hermes_cli.update_receipt import finalize_pending_update_receipt
|
||||
|
||||
_code = _update_exit.code if isinstance(_update_exit.code, int) else 1
|
||||
finalize_pending_update_receipt(_code, f"sys.exit({_code})")
|
||||
except Exception:
|
||||
pass
|
||||
raise
|
||||
except BaseException as _update_exc:
|
||||
try:
|
||||
from hermes_cli.update_receipt import finalize_pending_update_receipt
|
||||
|
||||
finalize_pending_update_receipt(
|
||||
1, f"{type(_update_exc).__name__}: {_update_exc}"
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
raise
|
||||
else:
|
||||
try:
|
||||
from hermes_cli.update_receipt import finalize_pending_update_receipt
|
||||
|
||||
finalize_pending_update_receipt(0, "completed at command boundary")
|
||||
except Exception:
|
||||
pass
|
||||
finally:
|
||||
_update_lock.release()
|
||||
_finalize_update_output(_update_io_state)
|
||||
|
||||
@@ -2948,7 +2948,11 @@ def list_authenticated_providers(
|
||||
|
||||
# Check if credentials exist
|
||||
has_creds = False
|
||||
if overlay.auth_type == "aws_sdk":
|
||||
if getattr(overlay, "keyless", False):
|
||||
# Keyless providers (opencode-free) are served anonymously —
|
||||
# there is no credential to check, so everyone is authenticated.
|
||||
has_creds = True
|
||||
elif overlay.auth_type == "aws_sdk":
|
||||
has_creds = _has_aws_sdk_creds_for_listing(hermes_slug)
|
||||
elif overlay.auth_type == "vertex":
|
||||
# Vertex authenticates via OAuth2 (service-account JSON / ADC),
|
||||
|
||||
+23
-12
@@ -135,11 +135,12 @@ OPENROUTER_MODELS: list[tuple[str, str]] = [
|
||||
# Free tier
|
||||
("stealth/ox-alpha", "free"), # "Ox Alpha" stealth reasoning model — 1M ctx
|
||||
("openrouter/elephant-alpha", "free"),
|
||||
("poolside/laguna-m.1:free", "free"),
|
||||
("tencent/hy3:free", "free"),
|
||||
("z-ai/glm-5.2:free", "free"),
|
||||
("poolside/laguna-s-2.1:free", "free"),
|
||||
("poolside/laguna-xs-2.1:free", "free"),
|
||||
("nvidia/nemotron-3-super-120b-a12b:free", "free"),
|
||||
("nvidia/nemotron-3-ultra-550b-a55b:free", "free"),
|
||||
("inclusionai/ring-2.6-1t:free", "free"),
|
||||
("nvidia/nemotron-3.5-lightning:free", "free"),
|
||||
]
|
||||
|
||||
_openrouter_catalog_cache: list[tuple[str, str]] | None = None
|
||||
@@ -519,7 +520,6 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
|
||||
"claude-opus-4-7",
|
||||
"claude-opus-4-6",
|
||||
"claude-opus-4-5",
|
||||
"claude-opus-4-1",
|
||||
"claude-sonnet-4-6",
|
||||
"claude-sonnet-4-5",
|
||||
"claude-sonnet-4",
|
||||
@@ -544,8 +544,6 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
|
||||
"deepseek-v4-pro",
|
||||
"deepseek-v4-flash",
|
||||
"deepseek-v4-flash-free",
|
||||
"qwen3.7-max",
|
||||
"qwen3.7-plus",
|
||||
"qwen3.6-plus",
|
||||
"qwen3.5-plus",
|
||||
"big-pickle",
|
||||
@@ -559,12 +557,14 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
|
||||
# OpenCode free tier — keyless (no OpenCode account needed). Synced
|
||||
# against live GET /zen/v1/models + anonymous probes (2026-08-21);
|
||||
# deepseek-v4-flash-free delisted (promo ended, now 401s).
|
||||
# big-pickle + mimo-v2.5-free delisted (UA-gated: the relay 429s
|
||||
# FreeUsageLimitError for every client except User-Agent
|
||||
# "opencode/latest"; we send honest Hermes attribution and don't
|
||||
# impersonate other clients — verified 2026-08-21).
|
||||
"opencode-free": [
|
||||
"x-preview-f-free", # "Ox Alpha" stealth model — free, 1M ctx, ZDR
|
||||
"big-pickle",
|
||||
"hy3-free",
|
||||
"laguna-s-2.1-free",
|
||||
"mimo-v2.5-free",
|
||||
"nemotron-3-ultra-free",
|
||||
"nemotron-3.5-lightning-free",
|
||||
"muse-spark-1.2-contributor-free",
|
||||
@@ -599,6 +599,9 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
|
||||
"hy3",
|
||||
"hy3-preview",
|
||||
"muse-spark-1.2-contributor",
|
||||
# Go-subscription twin of the Zen keyless Ox Alpha (live go/v1
|
||||
# catalog 2026-08-21; NOT keyless — Go relay requires a Go key).
|
||||
"ox-alpha-free",
|
||||
],
|
||||
"kilocode": [
|
||||
"anthropic/claude-opus-4.6",
|
||||
@@ -5399,14 +5402,22 @@ def opencode_zen_free_runtime(provider_id: Optional[str], model_id: Optional[str
|
||||
- ``provider_id`` is ``opencode-free`` (the dedicated keyless provider —
|
||||
EVERY model on it routes anonymously; that is the provider's contract), or
|
||||
- ``provider_id`` is any other OpenCode-family provider and ``model_id``
|
||||
is a free-tier slug (heals a free-model selection made under
|
||||
opencode-zen/opencode-go, whose keys the free tier rejects).
|
||||
is in the VERIFIED keyless catalog (``_PROVIDER_MODELS["opencode-free"]``)
|
||||
— heals a free-model selection made under opencode-zen/opencode-go,
|
||||
whose keys the free tier rejects.
|
||||
|
||||
Membership, not the ``-free`` suffix, is the heal criterion: the suffix
|
||||
stopped being a reliable keyless signal when ``ox-alpha-free`` appeared
|
||||
on the Go relay as a KEYED subscription model (2026-08-21) — suffix-based
|
||||
healing would have routed it to a Zen relay that doesn't serve it.
|
||||
"""
|
||||
family = opencode_provider_family(provider_id)
|
||||
if family is None:
|
||||
return None
|
||||
if family != "opencode-free" and not is_opencode_zen_free_model(model_id):
|
||||
return None
|
||||
if family != "opencode-free":
|
||||
bare = normalize_opencode_model_id(provider_id, model_id).strip().lower()
|
||||
if bare not in {m.lower() for m in _PROVIDER_MODELS.get("opencode-free", [])}:
|
||||
return None
|
||||
normalized = normalize_opencode_model_id(provider_id, model_id)
|
||||
api_mode = opencode_model_api_mode("opencode-zen", normalized)
|
||||
base_url = normalize_opencode_base_url(
|
||||
|
||||
+1
-1
@@ -109,7 +109,7 @@ _DEFAULT_PROVIDER_MODELS = {
|
||||
"ai-gateway": ["anthropic/claude-opus-4.6", "anthropic/claude-sonnet-4.6", "openai/gpt-5", "google/gemini-3-flash"],
|
||||
"kilocode": ["anthropic/claude-sonnet-5", "anthropic/claude-opus-4.6", "anthropic/claude-sonnet-4.6", "openai/gpt-5.4", "google/gemini-3-pro-preview", "google/gemini-3-flash-preview"],
|
||||
"opencode-zen": ["x-preview-f-free", "gpt-5.6-sol", "gpt-5.4", "gpt-5.3-codex", "claude-opus-5", "claude-sonnet-5", "gemini-3.7-flash", "glm-5.2", "kimi-k3", "minimax-m3"],
|
||||
"opencode-free": ["x-preview-f-free", "big-pickle", "hy3-free", "laguna-s-2.1-free", "mimo-v2.5-free", "nemotron-3-ultra-free", "nemotron-3.5-lightning-free", "muse-spark-1.2-contributor-free"],
|
||||
"opencode-free": ["x-preview-f-free", "hy3-free", "laguna-s-2.1-free", "nemotron-3-ultra-free", "nemotron-3.5-lightning-free", "muse-spark-1.2-contributor-free"],
|
||||
"opencode-go": ["kimi-k3", "kimi-k2.7-code", "kimi-k2.6", "gpt-5.6-luna", "grok-4.5", "glm-5.3", "glm-5.2", "mimo-v2.5-pro", "mimo-v2.5", "minimax-m3", "minimax-m2.7", "qwen3.8-max", "qwen3.7-max", "deepseek-v4-pro", "hy3"],
|
||||
"huggingface": [
|
||||
"Qwen/Qwen3.5-397B-A17B", "Qwen/Qwen3-235B-A22B-Thinking-2507",
|
||||
|
||||
@@ -31,6 +31,17 @@ def build_update_parser(subparsers, *, cmd_update: Callable) -> None:
|
||||
default=False,
|
||||
help="Check whether an update is available without installing anything",
|
||||
)
|
||||
update_parser.add_argument(
|
||||
"--plan",
|
||||
action="store_true",
|
||||
default=False,
|
||||
help=(
|
||||
"Show the update plan and exit without changing anything: install "
|
||||
"kind (git/docker/nix), every running Hermes service across all "
|
||||
"profiles with its supervisor and running code version, and how "
|
||||
"each will be restarted. Read-only; safe on a live fleet."
|
||||
),
|
||||
)
|
||||
update_parser.add_argument(
|
||||
"--no-backup",
|
||||
action="store_true",
|
||||
|
||||
+146
-23
@@ -4497,8 +4497,123 @@ def _warn_incomplete_gateway_fleet_restart(failed_units: list) -> None:
|
||||
print(" Skipped units may still be running pre-update code (mixed")
|
||||
print(" sys.modules). Restart them manually, then verify:")
|
||||
print(" hermes gateway status")
|
||||
print(" systemctl --user restart <unit> # user-scope")
|
||||
print(" sudo systemctl restart <unit> # system-scope")
|
||||
if any(not name.startswith("ai.hermes.") for name in ordered):
|
||||
print(" systemctl --user restart <unit> # user-scope")
|
||||
print(" sudo systemctl restart <unit> # system-scope")
|
||||
if any(name.startswith("ai.hermes.") for name in ordered):
|
||||
print(" launchctl kickstart -k gui/$UID/<label> # macOS (or user/$UID)")
|
||||
|
||||
|
||||
def _restart_macos_launchd_gateways(
|
||||
restarted_services: list,
|
||||
failed_or_stale_units: list,
|
||||
drain_budget: float,
|
||||
) -> None:
|
||||
"""Restart every launchd-managed gateway after an update (macOS).
|
||||
|
||||
The code update (git pull) is shared across all profiles, so every
|
||||
``ai.hermes.gateway*`` LaunchAgent must reload it — restarting only the
|
||||
invoking profile's service leaves siblings on pre-update ``sys.modules``
|
||||
until their next agent turn imports a symbol the old module generation
|
||||
doesn't have (#41403). Parity with the systemd fleet path.
|
||||
|
||||
The invoking profile keeps the existing ``launchd_restart()`` treatment
|
||||
(self-restart request → graceful drain → kickstart). Siblings get the
|
||||
same drain-first sequence, with their launchd domain resolved per label:
|
||||
a sibling bootstrapped in the other supported domain (``gui/<uid>`` vs
|
||||
``user/<uid>``) must not be kickstarted in the current profile's domain.
|
||||
``subprocess.TimeoutExpired`` is isolated per label so one wedged
|
||||
launchctl call cannot leave the rest of the fleet on old code (#68523).
|
||||
"""
|
||||
from hermes_cli.gateway import (
|
||||
get_launchd_label,
|
||||
get_launchd_plist_path,
|
||||
launchd_restart,
|
||||
launchd_gateway_labels_for_install,
|
||||
_graceful_restart_via_sigusr1,
|
||||
_launchd_kickstart,
|
||||
_launchd_service_registered,
|
||||
_locate_launchd_gateway_service,
|
||||
_wait_for_launchd_service_pid,
|
||||
)
|
||||
|
||||
# --- Current profile: unchanged single-service path ---------------------
|
||||
# Gate order and predicate mirror the pre-fleet inline block exactly:
|
||||
# plist first (no plist → zero launchctl calls), then the domain-agnostic
|
||||
# `launchctl list` registration check — NOT a domain locate, which fails
|
||||
# on macOS-26 hosts whose per-user domains reject service management
|
||||
# even though launchd_restart() owns that fallback. Gate errors skip
|
||||
# silently (best-effort, as before); only launchd_restart() itself
|
||||
# failing counts toward the incomplete-update warning.
|
||||
current_label = get_launchd_label()
|
||||
try:
|
||||
if get_launchd_plist_path().exists() and _launchd_service_registered(
|
||||
current_label
|
||||
):
|
||||
try:
|
||||
launchd_restart()
|
||||
restarted_services.append(current_label)
|
||||
except subprocess.CalledProcessError as e:
|
||||
stderr = (getattr(e, "stderr", "") or "").strip()
|
||||
print(f" ⚠ Gateway restart failed: {stderr}")
|
||||
failed_or_stale_units.append(current_label)
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
pass
|
||||
|
||||
# --- Sibling profiles ---------------------------------------------------
|
||||
for label in launchd_gateway_labels_for_install():
|
||||
if label == current_label:
|
||||
continue
|
||||
try:
|
||||
# Locate = liveness + domain in one domain-explicit probe; the
|
||||
# kickstart and fresh-PID verification below reuse the located
|
||||
# domain, so a sibling in the other gui/user domain can never be
|
||||
# probed in one domain and restarted in another.
|
||||
domain, old_pid = _locate_launchd_gateway_service(label)
|
||||
if domain is None:
|
||||
# Installed but not bootstrapped (stopped/uninstalled
|
||||
# mid-way) — nothing is running old code here.
|
||||
continue
|
||||
graceful_ok = False
|
||||
if old_pid is not None and old_pid > 0:
|
||||
print(f" → {label}: draining (up to {int(drain_budget)}s)...")
|
||||
graceful_ok = _graceful_restart_via_sigusr1(
|
||||
old_pid, drain_timeout=drain_budget
|
||||
)
|
||||
if graceful_ok and _wait_for_launchd_service_pid(
|
||||
label, old_pid=old_pid, timeout=10.0, domain=domain
|
||||
):
|
||||
# Unconditional KeepAlive already respawned it on the new
|
||||
# code — a hard kickstart now would kill the fresh process.
|
||||
restarted_services.append(label)
|
||||
continue
|
||||
try:
|
||||
_launchd_kickstart(label, domain)
|
||||
except subprocess.CalledProcessError as e:
|
||||
stderr = (getattr(e, "stderr", "") or "").strip()
|
||||
failed_or_stale_units.append(label)
|
||||
print(
|
||||
f" ⚠ Failed to restart {label}: {stderr}\n"
|
||||
f" Recover manually: launchctl kickstart -k {domain}/{label}"
|
||||
)
|
||||
continue
|
||||
if _wait_for_launchd_service_pid(
|
||||
label, old_pid=old_pid, timeout=15.0, domain=domain
|
||||
):
|
||||
restarted_services.append(label)
|
||||
else:
|
||||
failed_or_stale_units.append(label)
|
||||
print(
|
||||
f" ✗ {label} failed to come back after restart.\n"
|
||||
f" Check logs, then: launchctl kickstart -k {domain}/{label}"
|
||||
)
|
||||
except subprocess.TimeoutExpired:
|
||||
failed_or_stale_units.append(label)
|
||||
print(
|
||||
f" ⚠ launchctl timed out restarting {label}; "
|
||||
"continuing with remaining gateways"
|
||||
)
|
||||
|
||||
|
||||
def _surviving_gateway_pids_after_failed_restart():
|
||||
"""Best-effort PIDs of gateways still running after the restart phase died.
|
||||
@@ -5018,6 +5133,27 @@ def _cmd_update_impl(args, gateway_mode: bool):
|
||||
except Exception as _receipt_exc:
|
||||
logger.debug("Update receipt unavailable: %s", _receipt_exc)
|
||||
|
||||
# Plan phase (#91277 Phase 2): snapshot the pre-update fleet — every
|
||||
# running Hermes runtime, its supervisor, and its running code version —
|
||||
# into the receipt, so a post-mortem can compare what the update SAW
|
||||
# against what it did. Read-only; a probe failure records nothing.
|
||||
try:
|
||||
from hermes_cli.update_inventory import (
|
||||
collect_runtime_inventory,
|
||||
record_plan_in_receipt,
|
||||
)
|
||||
|
||||
_pre_update_plan = collect_runtime_inventory()
|
||||
record_plan_in_receipt(_pre_update_plan)
|
||||
if _pre_update_plan.runtimes:
|
||||
_n = len(_pre_update_plan.runtimes)
|
||||
_profiles = ", ".join(
|
||||
sorted({r.profile for r in _pre_update_plan.runtimes})
|
||||
)
|
||||
print(f"→ Fleet: {_n} running service(s) across profiles: {_profiles}")
|
||||
except Exception as _plan_exc:
|
||||
logger.debug("Update plan phase failed: %s", _plan_exc)
|
||||
|
||||
# On Windows, abort early if another hermes.exe is holding the venv shim
|
||||
# open. Continuing would result in a string of WinError 32 warnings and
|
||||
# then either a deferred-rename leftover or a failed git-pull fast path
|
||||
@@ -6966,30 +7102,17 @@ def _cmd_update_impl(args, gateway_mode: bool):
|
||||
)
|
||||
|
||||
# --- Launchd services (macOS) ---
|
||||
# Restart EVERY ai.hermes.gateway* LaunchAgent, not only the
|
||||
# invoking profile's — parity with the systemd branch above
|
||||
# (#41403). Per-label TimeoutExpired isolation happens inside.
|
||||
if is_macos():
|
||||
try:
|
||||
from hermes_cli.gateway import (
|
||||
launchd_restart,
|
||||
get_launchd_label,
|
||||
get_launchd_plist_path,
|
||||
_restart_macos_launchd_gateways(
|
||||
restarted_services,
|
||||
failed_or_stale_units,
|
||||
_drain_budget,
|
||||
)
|
||||
|
||||
plist_path = get_launchd_plist_path()
|
||||
if plist_path.exists():
|
||||
check = subprocess.run(
|
||||
["launchctl", "list", get_launchd_label()],
|
||||
capture_output=True,
|
||||
text=True, encoding="utf-8", errors="replace",
|
||||
timeout=5,
|
||||
)
|
||||
if check.returncode == 0:
|
||||
try:
|
||||
launchd_restart()
|
||||
restarted_services.append(get_launchd_label())
|
||||
except subprocess.CalledProcessError as e:
|
||||
stderr = (getattr(e, "stderr", "") or "").strip()
|
||||
print(f" ⚠ Gateway restart failed: {stderr}")
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired, ImportError):
|
||||
except (FileNotFoundError, ImportError):
|
||||
pass
|
||||
|
||||
# --- Manual (non-service) gateways ---
|
||||
|
||||
@@ -0,0 +1,274 @@
|
||||
"""Runtime inventory + update plan for the fleet-update pipeline (#91277 Phase 2).
|
||||
|
||||
One read-only pass that answers, BEFORE any mutation: what Hermes runtimes
|
||||
are running on this machine, how is each one deployed, which of them will
|
||||
this update touch, and how will each be restarted?
|
||||
|
||||
This is the "plan" phase of the transactional deployment model (#88683):
|
||||
|
||||
plan → snapshot → apply → restart-per-kind → verify → report
|
||||
|
||||
The module is deliberately side-effect free — every collector is a probe
|
||||
over primitives that already exist (`find_profile_gateway_processes`,
|
||||
`_get_service_pids`, `gateway_state.json` code stamps from #91283,
|
||||
`detect_install_method`) — so `hermes update --plan` can run on a live
|
||||
fleet with zero risk, and the update receipt can embed the inventory
|
||||
without changing update behavior.
|
||||
|
||||
Deployment kinds (the concept most fleet-update bugs were missing):
|
||||
|
||||
git — source checkout; updatable in place via `hermes update`
|
||||
docker — published image; NOT updatable in place (pull + recreate)
|
||||
nix/apt — package-manager owned; updatable via the manager only
|
||||
unknown — no marker; treated as in-place updatable (legacy default)
|
||||
|
||||
Supervisors (how a runtime is restarted after code changes):
|
||||
|
||||
systemd / launchd — restart via the service manager (fleet-wide)
|
||||
desktop — Desktop app supervises `hermes serve`; it respawns
|
||||
manual — plain process; SIGTERM + watcher/manual relaunch
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
from dataclasses import dataclass, field, asdict
|
||||
from pathlib import Path
|
||||
from typing import Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class RuntimeRecord:
|
||||
"""One running (or expected) Hermes runtime on this machine."""
|
||||
|
||||
kind: str # gateway | dashboard | serve
|
||||
profile: str # profile name ("default", ...)
|
||||
pid: Optional[int] = None # live PID when known
|
||||
supervisor: str = "manual" # systemd | launchd | desktop | manual
|
||||
code_sha: Optional[str] = None # stamped running-code sha (#91283)
|
||||
code_version: Optional[str] = None
|
||||
restart_via: str = "" # human-readable restart mechanism
|
||||
detail: dict = field(default_factory=dict)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return asdict(self)
|
||||
|
||||
|
||||
@dataclass
|
||||
class UpdatePlan:
|
||||
"""The full pre-update picture: install shape + runtimes + actions."""
|
||||
|
||||
install_method: str = "unknown" # git | docker | nix | apt | ...
|
||||
updatable_in_place: bool = True
|
||||
update_mechanism: str = "hermes update"
|
||||
expected_sha: Optional[str] = None # current checkout HEAD (pre-pull)
|
||||
expected_version: Optional[str] = None
|
||||
profiles: list = field(default_factory=list)
|
||||
runtimes: list = field(default_factory=list) # list[RuntimeRecord]
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
payload = asdict(self)
|
||||
payload["runtimes"] = [
|
||||
r.to_dict() if isinstance(r, RuntimeRecord) else r
|
||||
for r in self.runtimes
|
||||
]
|
||||
return payload
|
||||
|
||||
|
||||
def _detect_supervisor_for_pid(pid: int, service_pids: set) -> str:
|
||||
"""Classify how a live gateway PID is supervised."""
|
||||
if pid in service_pids:
|
||||
try:
|
||||
from hermes_cli.gateway import is_macos, supports_systemd_services
|
||||
|
||||
if supports_systemd_services():
|
||||
return "systemd"
|
||||
if is_macos():
|
||||
return "launchd"
|
||||
except Exception:
|
||||
pass
|
||||
return "service"
|
||||
return "manual"
|
||||
|
||||
|
||||
def _restart_mechanism(supervisor: str, profile: str) -> str:
|
||||
if supervisor == "systemd":
|
||||
return "systemctl restart (drain-first SIGUSR1 when supported)"
|
||||
if supervisor == "launchd":
|
||||
return "launchctl kickstart -k (drain-first, per-label domain)"
|
||||
if supervisor == "desktop":
|
||||
return "Desktop app respawns its serve backend"
|
||||
if profile != "default":
|
||||
return f"hermes -p {profile} gateway restart"
|
||||
return "hermes gateway restart"
|
||||
|
||||
|
||||
def collect_runtime_inventory() -> UpdatePlan:
|
||||
"""Build the pre-update plan. Read-only; never raises.
|
||||
|
||||
Every collector degrades independently — a probe failure yields fewer
|
||||
rows, not an exception. The result is embeddable in the update receipt
|
||||
and printable via :func:`print_update_plan`.
|
||||
"""
|
||||
plan = UpdatePlan()
|
||||
|
||||
# --- install shape / deployment kind ---------------------------------
|
||||
try:
|
||||
from hermes_cli.config import (
|
||||
detect_install_method,
|
||||
get_managed_system,
|
||||
recommended_update_command_for_method,
|
||||
)
|
||||
|
||||
method = detect_install_method()
|
||||
plan.install_method = method
|
||||
managed = get_managed_system()
|
||||
if managed:
|
||||
plan.install_method = managed
|
||||
plan.updatable_in_place = method in ("git", "unknown") and not managed
|
||||
plan.update_mechanism = recommended_update_command_for_method(method)
|
||||
except Exception as exc:
|
||||
logger.debug("Install-method probe failed: %s", exc)
|
||||
|
||||
# --- expected code identity (pre-pull) --------------------------------
|
||||
try:
|
||||
from hermes_cli.build_info import get_code_identity
|
||||
|
||||
identity = get_code_identity(refresh=True)
|
||||
plan.expected_sha = identity.get("sha")
|
||||
plan.expected_version = identity.get("version")
|
||||
except Exception as exc:
|
||||
logger.debug("Code-identity probe failed: %s", exc)
|
||||
|
||||
# --- profiles ----------------------------------------------------------
|
||||
profile_homes: list[tuple[str, Path]] = []
|
||||
try:
|
||||
from hermes_cli.profiles import (
|
||||
_get_default_hermes_home,
|
||||
_get_profiles_root,
|
||||
_PROFILE_ID_RE,
|
||||
)
|
||||
|
||||
default_home = _get_default_hermes_home()
|
||||
if default_home.is_dir():
|
||||
profile_homes.append(("default", default_home))
|
||||
root = _get_profiles_root()
|
||||
if root.is_dir():
|
||||
for entry in sorted(root.iterdir()):
|
||||
if (
|
||||
entry.is_dir()
|
||||
and entry.name != "default"
|
||||
and _PROFILE_ID_RE.match(entry.name)
|
||||
):
|
||||
profile_homes.append((entry.name, entry))
|
||||
plan.profiles = [name for name, _ in profile_homes]
|
||||
except Exception as exc:
|
||||
logger.debug("Profile enumeration failed: %s", exc)
|
||||
|
||||
# --- service-managed PIDs (fleet-wide) ---------------------------------
|
||||
service_pids: set = set()
|
||||
try:
|
||||
from hermes_cli.gateway import _get_service_pids
|
||||
|
||||
service_pids = _get_service_pids(all_profiles=True) or set()
|
||||
except Exception as exc:
|
||||
logger.debug("Service-PID probe failed: %s", exc)
|
||||
|
||||
# --- per-profile gateways (PID files + runtime status stamps) ----------
|
||||
seen_pids: set[int] = set()
|
||||
try:
|
||||
from gateway.status import _pid_exists, read_runtime_status
|
||||
|
||||
for profile, home in profile_homes:
|
||||
record = read_runtime_status(home / "gateway_state.json")
|
||||
pid: Optional[int] = None
|
||||
code_sha = code_version = None
|
||||
if record:
|
||||
try:
|
||||
pid = int(record.get("pid"))
|
||||
except (TypeError, ValueError):
|
||||
pid = None
|
||||
code_sha = record.get("code_sha")
|
||||
code_version = record.get("code_version")
|
||||
if pid is None or not _pid_exists(pid):
|
||||
continue
|
||||
seen_pids.add(pid)
|
||||
supervisor = _detect_supervisor_for_pid(pid, service_pids)
|
||||
plan.runtimes.append(
|
||||
RuntimeRecord(
|
||||
kind="gateway",
|
||||
profile=profile,
|
||||
pid=pid,
|
||||
supervisor=supervisor,
|
||||
code_sha=str(code_sha) if code_sha else None,
|
||||
code_version=code_version,
|
||||
restart_via=_restart_mechanism(supervisor, profile),
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("Gateway-state inventory failed: %s", exc)
|
||||
|
||||
# PID-file mapped gateways not covered by a runtime-status record
|
||||
try:
|
||||
from hermes_cli.gateway import find_profile_gateway_processes
|
||||
|
||||
for proc in find_profile_gateway_processes():
|
||||
if proc.pid in seen_pids:
|
||||
continue
|
||||
seen_pids.add(proc.pid)
|
||||
supervisor = _detect_supervisor_for_pid(proc.pid, service_pids)
|
||||
plan.runtimes.append(
|
||||
RuntimeRecord(
|
||||
kind="gateway",
|
||||
profile=proc.profile,
|
||||
pid=proc.pid,
|
||||
supervisor=supervisor,
|
||||
restart_via=_restart_mechanism(supervisor, proc.profile),
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("PID-file gateway inventory failed: %s", exc)
|
||||
|
||||
return plan
|
||||
|
||||
|
||||
def print_update_plan(plan: UpdatePlan) -> None:
|
||||
"""Human-readable plan — what the update will touch and how."""
|
||||
print("Update plan:")
|
||||
print(f" Install: {plan.install_method}", end="")
|
||||
if plan.expected_version:
|
||||
print(f" (v{plan.expected_version}", end="")
|
||||
if plan.expected_sha:
|
||||
print(f" @ {plan.expected_sha[:8]}", end="")
|
||||
print(")", end="")
|
||||
print()
|
||||
if not plan.updatable_in_place:
|
||||
print(" ⚠ This install is NOT updatable in place.")
|
||||
print(f" Update via: {plan.update_mechanism}")
|
||||
profiles = ", ".join(plan.profiles) if plan.profiles else "(none found)"
|
||||
print(f" Profiles: {profiles}")
|
||||
if not plan.runtimes:
|
||||
print(" Running Hermes services: none detected — code swap only.")
|
||||
return
|
||||
print(f" Running services to restart ({len(plan.runtimes)}):")
|
||||
for runtime in plan.runtimes:
|
||||
sha = f" @ {runtime.code_sha[:8]}" if runtime.code_sha else ""
|
||||
print(
|
||||
f" • {runtime.kind} [{runtime.profile}] pid {runtime.pid}"
|
||||
f" — {runtime.supervisor}{sha}"
|
||||
)
|
||||
print(f" restart: {runtime.restart_via}")
|
||||
|
||||
|
||||
def record_plan_in_receipt(plan: UpdatePlan) -> None:
|
||||
"""Attach the inventory to the active update receipt. Never raises."""
|
||||
try:
|
||||
import hermes_cli.update_receipt as ur
|
||||
|
||||
if ur._current is not None:
|
||||
ur._current.data["plan"] = plan.to_dict()
|
||||
except Exception as exc:
|
||||
logger.debug("Could not record plan in receipt: %s", exc)
|
||||
@@ -166,10 +166,15 @@ def record_gateway_restart(**kwargs: Any) -> None:
|
||||
logger.debug("Could not record gateway restart result: %s", exc)
|
||||
|
||||
|
||||
def finalize_update_receipt(outcome: str, fleet: list | None = None) -> Optional[Path]:
|
||||
def finalize_update_receipt(
|
||||
outcome: str, fleet: list | None = None, stop_reason: str = ""
|
||||
) -> Optional[Path]:
|
||||
"""Finalize + persist the receipt. Returns the written path or None.
|
||||
|
||||
``outcome`` is one of ``success`` / ``partial`` / ``failed``.
|
||||
``outcome`` is one of ``success`` / ``partial`` / ``failed`` /
|
||||
``refused``. Exactly-once by construction: the module singleton is
|
||||
popped first, so a second call (e.g. the command-boundary safety net
|
||||
after an inner path already finalized) is a no-op returning None.
|
||||
"""
|
||||
global _current
|
||||
receipt = _current
|
||||
@@ -178,6 +183,8 @@ def finalize_update_receipt(outcome: str, fleet: list | None = None) -> Optional
|
||||
return None
|
||||
try:
|
||||
receipt.finalize(outcome)
|
||||
if stop_reason:
|
||||
receipt.data["stop_reason"] = stop_reason
|
||||
if fleet is not None:
|
||||
receipt.data["fleet"] = fleet
|
||||
directory = _receipt_dir()
|
||||
@@ -202,6 +209,42 @@ def finalize_update_receipt(outcome: str, fleet: list | None = None) -> Optional
|
||||
return None
|
||||
|
||||
|
||||
def finalize_pending_update_receipt(
|
||||
exit_code: Optional[int] = None, stop_reason: str = ""
|
||||
) -> Optional[Path]:
|
||||
"""Command-boundary safety net: persist a still-open receipt, if any.
|
||||
|
||||
``hermes update`` has many early-termination paths (Windows
|
||||
concurrent-instance preflight, venv-holder refusal, head-pinned no-op,
|
||||
fetch failure — all ``sys.exit``) that predate the inner finalize
|
||||
call sites. Any receipt still open when the update COMMAND unwinds is
|
||||
finalized here so every post-begin run leaves a record — the
|
||||
refused/failed runs are exactly the ones a receipt matters most for
|
||||
(review on #91283). No-op when no receipt is open (the inner paths
|
||||
already finalized — exactly-once via the popped singleton) or when
|
||||
recording was never started. Never raises.
|
||||
|
||||
Outcome mapping: exit 0/None → ``success`` (a path that completed
|
||||
without an explicit inner finalize), exit 2 → ``refused`` (the
|
||||
updater's preflight-refusal convention), anything else → ``failed``.
|
||||
"""
|
||||
if _current is None:
|
||||
return None
|
||||
if exit_code in (0, None):
|
||||
outcome = "success"
|
||||
elif exit_code == 2:
|
||||
outcome = "refused"
|
||||
else:
|
||||
outcome = "failed"
|
||||
try:
|
||||
receipt = _current
|
||||
if receipt is not None and exit_code is not None:
|
||||
receipt.data["exit_code"] = int(exit_code)
|
||||
except Exception:
|
||||
pass
|
||||
return finalize_update_receipt(outcome, stop_reason=stop_reason)
|
||||
|
||||
|
||||
def _prune_old_receipts(directory: Path) -> None:
|
||||
try:
|
||||
receipts = sorted(
|
||||
|
||||
@@ -9,6 +9,8 @@ hermes_cli.models.opencode_zen_free_runtime). No OpenCode account needed.
|
||||
Select via ``hermes model`` or ``/model free``.
|
||||
"""
|
||||
|
||||
from typing import Any
|
||||
|
||||
from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
@@ -23,7 +25,33 @@ _KEYLESS_HEADERS = {
|
||||
"User-Agent": f"HermesAgent/{_HERMES_VERSION}",
|
||||
}
|
||||
|
||||
opencode_free = ProviderProfile(
|
||||
|
||||
class OpenCodeFreeProfile(ProviderProfile):
|
||||
"""OpenCode Free — keyless, with Ox Alpha reasoning controls.
|
||||
|
||||
Ox Alpha (x-preview-f-free) is reachable through this provider as well
|
||||
as opencode-zen; both share the same wire contract (reasoning_effort
|
||||
accepts exactly low/high/max — anything else 400s). The translation
|
||||
lives in the zen plugin; resolve it through the registered zen profile's
|
||||
module so the two providers can never drift.
|
||||
"""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
try:
|
||||
import sys
|
||||
|
||||
from providers import get_provider_profile
|
||||
|
||||
zen_profile = get_provider_profile("opencode-zen")
|
||||
zen_module = sys.modules[type(zen_profile).__module__]
|
||||
return zen_module._build_ox_alpha_reasoning_extras(reasoning_config, model)
|
||||
except Exception:
|
||||
return {}, {}
|
||||
|
||||
|
||||
opencode_free = OpenCodeFreeProfile(
|
||||
name="opencode-free",
|
||||
aliases=("free", "opencode_free"),
|
||||
env_vars=(), # keyless — nothing to configure
|
||||
@@ -31,7 +59,9 @@ opencode_free = ProviderProfile(
|
||||
display_name="OpenCode Free",
|
||||
description="OpenCode free models — keyless, no account needed",
|
||||
default_headers=dict(_KEYLESS_HEADERS),
|
||||
default_aux_model="big-pickle",
|
||||
# laguna is the fastest non-UA-gated free model; big-pickle 429s every
|
||||
# client except the opencode CLI's own User-Agent (verified 2026-08-21).
|
||||
default_aux_model="laguna-s-2.1-free",
|
||||
)
|
||||
|
||||
register_provider(opencode_free)
|
||||
|
||||
@@ -158,7 +158,48 @@ class OpenCodeGoProfile(ProviderProfile):
|
||||
return extra_body, top_level
|
||||
|
||||
|
||||
opencode_zen = ProviderProfile(
|
||||
def _build_ox_alpha_reasoning_extras(
|
||||
reasoning_config: dict | None, model: str | None
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Shared Ox Alpha (x-preview-f-free) reasoning_effort translation.
|
||||
|
||||
Used by both the opencode-zen profile and the opencode-free keyless
|
||||
profile — the model is reachable through either provider and the wire
|
||||
contract is identical (low/high/max only; anything else 400s).
|
||||
"""
|
||||
if _flat_model_name(model) != "x-preview-f-free":
|
||||
return {}, {}
|
||||
if not isinstance(reasoning_config, dict):
|
||||
return {}, {}
|
||||
if reasoning_config.get("enabled") is False:
|
||||
return {}, {}
|
||||
|
||||
effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
if not effort or effort == "none":
|
||||
return {}, {}
|
||||
|
||||
from agent.reasoning_effort import (
|
||||
OX_ALPHA_EFFORTS,
|
||||
OX_ALPHA_OVERRIDES,
|
||||
clamp_effort,
|
||||
)
|
||||
|
||||
clamped = clamp_effort(effort, OX_ALPHA_EFFORTS, OX_ALPHA_OVERRIDES)
|
||||
if clamped not in OX_ALPHA_EFFORTS:
|
||||
return {}, {}
|
||||
return {}, {"reasoning_effort": clamped}
|
||||
|
||||
|
||||
class OpenCodeZenProfile(ProviderProfile):
|
||||
"""OpenCode Zen - model-specific reasoning controls."""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
return _build_ox_alpha_reasoning_extras(reasoning_config, model)
|
||||
|
||||
|
||||
opencode_zen = OpenCodeZenProfile(
|
||||
name="opencode-zen",
|
||||
aliases=("opencode", "opencode_zen", "zen"),
|
||||
env_vars=("OPENCODE_ZEN_API_KEY",),
|
||||
|
||||
+1
-1
@@ -2,7 +2,7 @@
|
||||
|
||||
[project]
|
||||
name = "hermes-agent"
|
||||
version = "0.20.4"
|
||||
version = "0.20.5"
|
||||
description = "The self-improving AI agent — creates skills from experience, improves them during use, and runs anywhere"
|
||||
readme = "README.md"
|
||||
# Upper bound is load-bearing, not cosmetic. uv resolves the project's
|
||||
|
||||
+39
-15
@@ -1897,22 +1897,37 @@ class AIAgent:
|
||||
enabled, task_cfg = load_background_review_settings()
|
||||
if not enabled:
|
||||
return
|
||||
from agent.background_review import spawn_background_review_thread
|
||||
from agent.background_review import (
|
||||
finish_background_review_run,
|
||||
prepare_background_review_run,
|
||||
spawn_background_review_thread,
|
||||
)
|
||||
from tools.thread_context import propagate_context_to_thread
|
||||
target, _prompt = spawn_background_review_thread(
|
||||
self,
|
||||
messages_snapshot,
|
||||
review_memory=review_memory,
|
||||
review_skills=review_skills,
|
||||
focus=focus,
|
||||
task_cfg=task_cfg,
|
||||
)
|
||||
# Carry the active profile into the review thread so MEMORY.md / skill
|
||||
# review writes land in the right profile (#54937).
|
||||
t = threading.Thread(
|
||||
target=propagate_context_to_thread(target), daemon=True, name="bg-review"
|
||||
)
|
||||
t.start()
|
||||
|
||||
review_run = prepare_background_review_run(self)
|
||||
if review_run is None:
|
||||
return
|
||||
try:
|
||||
target, _prompt = spawn_background_review_thread(
|
||||
self,
|
||||
messages_snapshot,
|
||||
review_memory=review_memory,
|
||||
review_skills=review_skills,
|
||||
focus=focus,
|
||||
task_cfg=task_cfg,
|
||||
review_run=review_run,
|
||||
)
|
||||
# Carry the active profile into the review thread so MEMORY.md /
|
||||
# skill review writes land in the right profile (#54937).
|
||||
t = threading.Thread(
|
||||
target=propagate_context_to_thread(target),
|
||||
daemon=True,
|
||||
name="bg-review",
|
||||
)
|
||||
t.start()
|
||||
except Exception:
|
||||
finish_background_review_run(self, review_run)
|
||||
raise
|
||||
|
||||
def _build_memory_write_metadata(
|
||||
self,
|
||||
@@ -8480,6 +8495,15 @@ class AIAgent:
|
||||
moa_config: Optional[dict[str, Any]] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Forwarder — see ``agent.conversation_loop.run_conversation``."""
|
||||
# A review deliberately shares this agent's session_id for prompt-cache
|
||||
# parity. Fence review startup or interrupt an admitted request, then
|
||||
# await that request's exit before opening any live-turn Relay or task
|
||||
# instrumentation for the same session. Foreground priority is retained
|
||||
# if the review does not acknowledge within the bounded deadline (#84423).
|
||||
from agent.background_review import cancel_background_review_for_live_turn
|
||||
|
||||
cancel_background_review_for_live_turn(self)
|
||||
|
||||
from agent.aux_accounting import (
|
||||
reset_accounting_context,
|
||||
set_accounting_context,
|
||||
|
||||
@@ -571,6 +571,59 @@ function Start-DesktopRelaunch {
|
||||
} catch {
|
||||
Write-HandoffLog "WARNING: WMI relaunch failed: $($_.Exception.Message); falling back"
|
||||
}
|
||||
if (-not $spawned) {
|
||||
# Middle rung: explorer.exe-mediated launch. On some machines
|
||||
# Win32_Process.Create fails outright (observed ReturnValue 8,
|
||||
# "unknown failure"), and the tethered fallback below re-attaches the
|
||||
# Desktop to this console — its stdout then floods the console and the
|
||||
# window can't close while the app lives. Explorer re-parents the
|
||||
# target exactly like a normal shell launch, giving the same
|
||||
# no-console detachment WMI would have. Explorer returns no pid, so
|
||||
# verify by watching for a fresh Hermes process.
|
||||
try {
|
||||
$exeName = [System.IO.Path]::GetFileNameWithoutExtension($RelaunchExe)
|
||||
$before = @(Get-Process -Name $exeName -ErrorAction SilentlyContinue | ForEach-Object { $_.Id })
|
||||
Start-Process -FilePath 'explorer.exe' -ArgumentList ('"{0}"' -f $RelaunchExe) | Out-Null
|
||||
$explorerDeadline = (Get-Date).AddSeconds(15)
|
||||
while ((Get-Date) -lt $explorerDeadline) {
|
||||
$fresh = @(Get-Process -Name $exeName -ErrorAction SilentlyContinue | Where-Object { $before -notcontains $_.Id })
|
||||
if ($fresh.Count -gt 0) {
|
||||
Write-HandoffLog "desktop relaunched detached via explorer (pid $($fresh[0].Id))"
|
||||
$spawned = $true
|
||||
# Same foreground hand-off as the WMI rung: the new process
|
||||
# starts unfocused and only the current foreground owner
|
||||
# (us) can delegate that right.
|
||||
try {
|
||||
if ($script:Win32) {
|
||||
[HermesHandoff.Win32]::AllowSetForegroundWindow([int]$fresh[0].Id) | Out-Null
|
||||
$focusDeadline = (Get-Date).AddSeconds(20)
|
||||
while ((Get-Date) -lt $focusDeadline) {
|
||||
$hwnd = [System.IntPtr]::Zero
|
||||
try { $hwnd = (Get-Process -Id $fresh[0].Id -ErrorAction Stop).MainWindowHandle } catch { break }
|
||||
if ($hwnd -ne [System.IntPtr]::Zero) {
|
||||
[HermesHandoff.Win32]::ShowWindow($hwnd, 9) | Out-Null # SW_RESTORE
|
||||
[HermesHandoff.Win32]::SetForegroundWindow($hwnd) | Out-Null
|
||||
Write-HandoffLog "focused relaunched desktop window"
|
||||
break
|
||||
}
|
||||
Start-Sleep -Milliseconds 400
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
Write-HandoffLog "WARNING: could not focus relaunched desktop: $($_.Exception.Message)"
|
||||
}
|
||||
break
|
||||
}
|
||||
Start-Sleep -Milliseconds 400
|
||||
if ($script:Ui) { [System.Windows.Forms.Application]::DoEvents() }
|
||||
}
|
||||
if (-not $spawned) {
|
||||
Write-HandoffLog "WARNING: explorer relaunch did not produce a $exeName process; falling back"
|
||||
}
|
||||
} catch {
|
||||
Write-HandoffLog "WARNING: explorer relaunch failed: $($_.Exception.Message); falling back"
|
||||
}
|
||||
}
|
||||
if (-not $spawned) {
|
||||
try {
|
||||
# Fallback keeps the old behavior (console tie-in and all) --
|
||||
|
||||
@@ -24,7 +24,9 @@ class _FakeOpenAI:
|
||||
pass
|
||||
|
||||
|
||||
def _make_agent(monkeypatch, enabled_toolsets=None, skip_memory=True):
|
||||
def _make_agent(
|
||||
monkeypatch, enabled_toolsets=None, disabled_toolsets=None, skip_memory=True
|
||||
):
|
||||
monkeypatch.setattr("run_agent.get_tool_definitions", lambda **kw: [])
|
||||
monkeypatch.setattr("run_agent.check_toolset_requirements", lambda: {})
|
||||
monkeypatch.setattr("run_agent.OpenAI", _FakeOpenAI)
|
||||
@@ -38,6 +40,7 @@ def _make_agent(monkeypatch, enabled_toolsets=None, skip_memory=True):
|
||||
skip_context_files=True,
|
||||
skip_memory=skip_memory,
|
||||
enabled_toolsets=enabled_toolsets,
|
||||
disabled_toolsets=disabled_toolsets,
|
||||
)
|
||||
|
||||
|
||||
@@ -96,3 +99,36 @@ def test_skip_memory_memory_tool_handler_works_and_provider_skipped(
|
||||
memory_md = tmp_path / "hm" / "memories" / "MEMORY.md"
|
||||
assert memory_md.exists()
|
||||
assert "User prefers concise answers." in memory_md.read_text()
|
||||
|
||||
|
||||
def test_skip_memory_disabled_toolset_does_not_load_store(monkeypatch, tmp_path):
|
||||
"""Cron shape: skip_memory=True, memory named in enabled AND disabled.
|
||||
|
||||
#65429 must not load MEMORY.md just because the default cron toolset
|
||||
still lists memory while the denylist hides the tool.
|
||||
"""
|
||||
home = tmp_path / "hm"
|
||||
monkeypatch.setenv("HERMES_HOME", str(home))
|
||||
mem_dir = home / "memories"
|
||||
mem_dir.mkdir(parents=True)
|
||||
secret = "cron-should-never-see-this-memory"
|
||||
(mem_dir / "MEMORY.md").write_text(secret + "\n")
|
||||
(mem_dir / "USER.md").write_text("cron-should-never-see-this-profile\n")
|
||||
|
||||
agent = _make_agent(
|
||||
monkeypatch,
|
||||
enabled_toolsets=["memory", "file"],
|
||||
disabled_toolsets=["memory"],
|
||||
skip_memory=True,
|
||||
)
|
||||
assert agent._memory_store is None
|
||||
assert agent._memory_manager is None
|
||||
assert agent._memory_enabled is False
|
||||
assert agent._user_profile_enabled is False
|
||||
|
||||
from agent.system_prompt import build_system_prompt_parts
|
||||
|
||||
parts = build_system_prompt_parts(agent)
|
||||
blob = " ".join(str(v) for v in parts.values())
|
||||
assert secret not in blob
|
||||
assert "cron-should-never-see-this-profile" not in blob
|
||||
|
||||
@@ -7,8 +7,8 @@ default off) makes that denial opt-out-able:
|
||||
|
||||
- gate off / absent: byte-exact current behavior — ``cronjob`` denied.
|
||||
- gate on: ``cronjob`` dropped from the base denylist; ``messaging`` and
|
||||
``clarify`` (interactivity constraints) and ``memory`` (cron agents run
|
||||
with skip_memory=True) are ALWAYS denied regardless of the gate.
|
||||
``clarify`` (interactivity constraints) are ALWAYS denied regardless of
|
||||
the gate.
|
||||
- user-level ``agent.disabled_toolsets`` still layers on top, so a user who
|
||||
denies ``cronjob`` globally keeps it denied even with the gate on
|
||||
(per-job enabled_toolsets can never widen past the config denylist).
|
||||
@@ -20,26 +20,27 @@ from cron.scheduler import _resolve_cron_disabled_toolsets
|
||||
|
||||
|
||||
# The toolsets that must be denied in cron context no matter what the
|
||||
# agent-scheduling gate says: messaging/clarify are interactive-only,
|
||||
# memory is unbacked in cron runs (skip_memory=True).
|
||||
ALWAYS_DISABLED = ["messaging", "clarify", "memory"]
|
||||
# agent-scheduling gate says: messaging/clarify are interactive-only.
|
||||
# ``memory`` is intentionally NOT here — cron agents get memory like any
|
||||
# other agent run.
|
||||
ALWAYS_DISABLED = ["messaging", "clarify"]
|
||||
|
||||
|
||||
class TestGateOffDefault:
|
||||
def test_empty_config_denies_cronjob(self):
|
||||
assert _resolve_cron_disabled_toolsets({}) == [
|
||||
"cronjob", "messaging", "clarify", "memory",
|
||||
"cronjob", "messaging", "clarify",
|
||||
]
|
||||
|
||||
def test_none_config_denies_cronjob(self):
|
||||
assert _resolve_cron_disabled_toolsets(None) == [
|
||||
"cronjob", "messaging", "clarify", "memory",
|
||||
"cronjob", "messaging", "clarify",
|
||||
]
|
||||
|
||||
def test_cron_section_present_but_gate_absent(self):
|
||||
cfg = {"cron": {"preflight": True}}
|
||||
assert _resolve_cron_disabled_toolsets(cfg) == [
|
||||
"cronjob", "messaging", "clarify", "memory",
|
||||
"cronjob", "messaging", "clarify",
|
||||
]
|
||||
|
||||
def test_explicit_false_matches_default(self):
|
||||
@@ -60,12 +61,18 @@ class TestGateOn:
|
||||
disabled = _resolve_cron_disabled_toolsets(cfg)
|
||||
assert "cronjob" not in disabled
|
||||
|
||||
def test_interactivity_and_memory_denials_survive_the_gate(self):
|
||||
def test_interactivity_denials_survive_the_gate(self):
|
||||
cfg = {"cron": {"allow_agent_scheduling": True}}
|
||||
disabled = _resolve_cron_disabled_toolsets(cfg)
|
||||
for name in ALWAYS_DISABLED:
|
||||
assert name in disabled
|
||||
|
||||
def test_memory_not_denied(self):
|
||||
# Cron agents run with memory enabled like any other agent run
|
||||
# (skip_memory=False); the toolset must not be policy-denied.
|
||||
for cfg in ({}, {"cron": {"allow_agent_scheduling": True}}):
|
||||
assert "memory" not in _resolve_cron_disabled_toolsets(cfg)
|
||||
|
||||
def test_user_denylist_wins_over_gate(self):
|
||||
# A user who denies cronjob in agent.disabled_toolsets keeps it
|
||||
# denied even with the gate on — the gate only removes the built-in
|
||||
@@ -94,6 +101,12 @@ class TestUserLayerUnchanged:
|
||||
# No duplicate when the user names an already-denied toolset.
|
||||
assert disabled.count("cronjob") == 1
|
||||
|
||||
def test_user_can_still_deny_memory_for_cron(self):
|
||||
# Memory is no longer policy-denied, but a user-level denylist
|
||||
# entry still applies to cron runs.
|
||||
cfg = {"agent": {"disabled_toolsets": ["memory"]}}
|
||||
assert "memory" in _resolve_cron_disabled_toolsets(cfg)
|
||||
|
||||
def test_blank_and_whitespace_entries_ignored(self):
|
||||
cfg = {
|
||||
"cron": {"allow_agent_scheduling": True},
|
||||
|
||||
@@ -118,6 +118,23 @@ class TestPerJobToolsetMcpMerge:
|
||||
assert m_platform.call_args[0][1] == "cron"
|
||||
assert set(result) == set(sentinel)
|
||||
|
||||
def test_resolver_keeps_memory_in_per_job_list(self):
|
||||
result = _resolve_cron_enabled_toolsets(
|
||||
{"enabled_toolsets": ["memory", "file"]},
|
||||
{"mcp_servers": {}},
|
||||
)
|
||||
assert "memory" in result
|
||||
assert "file" in result
|
||||
|
||||
def test_resolver_keeps_memory_from_platform_fallback(self):
|
||||
job = {"enabled_toolsets": None}
|
||||
with patch(
|
||||
"hermes_cli.tools_config._get_platform_tools",
|
||||
return_value={"web", "memory", "file"},
|
||||
):
|
||||
result = _resolve_cron_enabled_toolsets(job, {})
|
||||
assert result == ["file", "memory", "web"]
|
||||
|
||||
|
||||
class TestResolveOrigin:
|
||||
def test_full_origin(self):
|
||||
@@ -610,16 +627,15 @@ class TestRunJobSessionPersistence:
|
||||
yield fake_db, mock_agent_cls
|
||||
|
||||
|
||||
def test_run_job_memory_toolset_disabled_in_cron(self, tmp_path):
|
||||
"""memory toolset must be disabled in cron sessions — issue #38129.
|
||||
def test_run_job_memory_enabled_in_cron(self, tmp_path):
|
||||
"""Cron agents get memory like any other agent run.
|
||||
|
||||
Cron agents are constructed with skip_memory=True, so the memory
|
||||
backend is not initialised. Exposing the memory tool only gives the
|
||||
model an unbacked tool that fails at runtime with
|
||||
"Memory is not available." Hiding it from the schema prevents that.
|
||||
skip_memory=False and the memory toolset is not policy-denied, so
|
||||
MEMORY.md/USER.md load and the memory tool follows normal toolset
|
||||
resolution.
|
||||
"""
|
||||
job = {
|
||||
"id": "memory-hide-job",
|
||||
"id": "memory-enabled-job",
|
||||
"name": "test",
|
||||
"prompt": "hello",
|
||||
}
|
||||
@@ -627,18 +643,13 @@ class TestRunJobSessionPersistence:
|
||||
run_job(job)
|
||||
|
||||
kwargs = mock_agent_cls.call_args.kwargs
|
||||
assert "memory" in (kwargs["disabled_toolsets"] or []), (
|
||||
"memory toolset should be disabled in cron to match skip_memory=True"
|
||||
assert kwargs["skip_memory"] is False
|
||||
assert "memory" not in (kwargs["disabled_toolsets"] or []), (
|
||||
"memory toolset must not be policy-denied in cron"
|
||||
)
|
||||
|
||||
def test_run_job_disables_memory_even_when_per_job_enables_it(self, tmp_path):
|
||||
"""Cron runs pass skip_memory=True, so memory must not be exposed.
|
||||
|
||||
A cron job can request the memory tool through enabled_toolsets, but
|
||||
there is no MemoryStore injected for cron agents. Keep memory in the
|
||||
disabled set so AIAgent filters the unbacked tool out before the model
|
||||
can call it and receive "Memory is not available" failures.
|
||||
"""
|
||||
def test_run_job_keeps_per_job_memory_toolset(self, tmp_path):
|
||||
"""A per-job enabled_toolsets naming memory keeps it."""
|
||||
job = {
|
||||
"id": "memory-toolset-job",
|
||||
"name": "test",
|
||||
@@ -649,9 +660,10 @@ class TestRunJobSessionPersistence:
|
||||
run_job(job)
|
||||
|
||||
kwargs = mock_agent_cls.call_args.kwargs
|
||||
assert kwargs["skip_memory"] is True
|
||||
assert kwargs["enabled_toolsets"] == ["memory", "file"]
|
||||
assert "memory" in kwargs["disabled_toolsets"]
|
||||
assert kwargs["skip_memory"] is False
|
||||
assert "memory" in (kwargs["enabled_toolsets"] or [])
|
||||
assert "file" in (kwargs["enabled_toolsets"] or [])
|
||||
assert "memory" not in kwargs["disabled_toolsets"]
|
||||
|
||||
def test_tick_skips_due_jobs_while_dispatch_is_paused(self, tmp_path):
|
||||
"""The drain gate runs before advancing a due job's schedule."""
|
||||
|
||||
@@ -36,26 +36,6 @@ if _repo not in sys.path:
|
||||
sys.path.insert(0, _repo)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# python-telegram-bot is an optional dep; mock it so the adapter imports
|
||||
# (same shim as test_telegram_network_reconnect / test_telegram_plugin_handlers).
|
||||
# ---------------------------------------------------------------------------
|
||||
def _ensure_telegram_mock() -> None:
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
from gateway.run import GatewayRunner # noqa: E402
|
||||
from gateway.profile_routing import ProfileRoute # noqa: E402
|
||||
|
||||
@@ -42,27 +42,6 @@ class TestExtractMediaImages:
|
||||
# ---------------------------------------------------------------------------
|
||||
# Telegram send_image_file tests
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
"""Install mock telegram modules so TelegramAdapter can be imported."""
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -88,27 +88,6 @@ class TestBaseDefaultLoop:
|
||||
assert a.sent_files[0][1] == "/tmp/foo.png"
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Telegram mocks setup (shared with test_send_image_file pattern)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -16,36 +16,6 @@ if _repo not in sys.path:
|
||||
sys.path.insert(0, _repo)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Minimal Telegram mock so TelegramAdapter can be imported
|
||||
# ---------------------------------------------------------------------------
|
||||
def _ensure_telegram_mock():
|
||||
"""Wire up the minimal mocks required to import TelegramAdapter."""
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
mod = MagicMock()
|
||||
mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
mod.constants.ParseMode.MARKDOWN = "Markdown"
|
||||
mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
mod.constants.ParseMode.HTML = "HTML"
|
||||
mod.constants.ChatType.PRIVATE = "private"
|
||||
mod.constants.ChatType.GROUP = "group"
|
||||
mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
mod.constants.ChatType.CHANNEL = "channel"
|
||||
# Provide real exception classes so ``except (NetworkError, ...)`` in
|
||||
# connect() doesn't blow up under xdist when this mock leaks.
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter
|
||||
from gateway.config import Platform, PlatformConfig
|
||||
|
||||
|
||||
@@ -23,30 +23,6 @@ if _repo not in sys.path:
|
||||
# Minimal Telegram mock so TelegramAdapter can be imported (mirrors
|
||||
# test_telegram_approval_buttons.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
mod = MagicMock()
|
||||
mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
mod.constants.ParseMode.MARKDOWN = "Markdown"
|
||||
mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
mod.constants.ParseMode.HTML = "HTML"
|
||||
mod.constants.ChatType.PRIVATE = "private"
|
||||
mod.constants.ChatType.GROUP = "group"
|
||||
mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
mod.constants.ChatType.CHANNEL = "channel"
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
@@ -26,31 +26,12 @@ client-level limits when a custom transport is supplied.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import sys
|
||||
from unittest.mock import MagicMock, patch
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram import adapter as tg_adapter # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
@@ -1,39 +1,10 @@
|
||||
import asyncio
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
|
||||
# Provide real exception classes so ``except (NetworkError, ...)`` in
|
||||
# connect() doesn't blow up with "catching classes that do not inherit
|
||||
# from BaseException" when another xdist worker pollutes sys.modules.
|
||||
telegram_mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
telegram_mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
telegram_mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
sys.modules.setdefault("telegram.error", telegram_mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -5,37 +5,10 @@ must set a non-retryable fatal error so the gateway does not queue it for
|
||||
background reconnection (#31049).
|
||||
"""
|
||||
|
||||
import sys
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
|
||||
telegram_mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
telegram_mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
telegram_mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
sys.modules.setdefault("telegram.error", telegram_mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
import plugins.platforms.telegram.adapter as telegram_mod # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
@@ -10,7 +10,6 @@ We mock the telegram module at import time to avoid collection errors.
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
@@ -28,28 +27,6 @@ from gateway.platforms.base import (
|
||||
# ---------------------------------------------------------------------------
|
||||
# Mock the telegram package if it's not installed
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
"""Install mock telegram modules so TelegramAdapter can be imported."""
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
# Real library is installed — no mocking needed
|
||||
return
|
||||
|
||||
telegram_mod = MagicMock()
|
||||
# ContextTypes needs DEFAULT_TYPE as an actual attribute for the annotation
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
# Now we can safely import
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
@@ -6,7 +6,6 @@ or corrupt user-visible content.
|
||||
"""
|
||||
|
||||
import re
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
@@ -18,23 +17,6 @@ from gateway.config import PlatformConfig
|
||||
# ---------------------------------------------------------------------------
|
||||
# Mock the telegram package if it's not installed
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
mod = MagicMock()
|
||||
mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
mod.constants.ChatType.GROUP = "group"
|
||||
mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
mod.constants.ChatType.CHANNEL = "channel"
|
||||
mod.constants.ChatType.PRIVATE = "private"
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import ( # noqa: E402
|
||||
TelegramAdapter,
|
||||
_escape_mdv2,
|
||||
|
||||
@@ -1,30 +1,6 @@
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
telegram_mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
telegram_mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
sys.modules.setdefault("telegram.error", telegram_mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram import adapter as tg_adapter # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
@@ -5,30 +5,8 @@ The public Telegram Bot API caps `getFile` at 20MB. A locally-hosted
|
||||
of `extra.base_url` as the explicit opt-in to the higher cap.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -15,38 +15,9 @@ that actually reaches the Bot API.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
"""Install mock telegram modules only when the real library is absent.
|
||||
|
||||
Probe with a real import rather than inspecting ``sys.modules``: when PTB
|
||||
is installed but not yet imported, a ``sys.modules`` check alone would
|
||||
install the mock first and poison every later test that needs the real
|
||||
library (``tests/test_telegram_polling_progress_ptb.py`` exercises the
|
||||
genuine ``BaseRequest``).
|
||||
"""
|
||||
try:
|
||||
import telegram # noqa: F401
|
||||
|
||||
return
|
||||
except ImportError:
|
||||
pass
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
for name in ("GROUP", "SUPERGROUP", "CHANNEL", "PRIVATE"):
|
||||
setattr(telegram_mod.constants.ChatType, name, name.lower())
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from gateway.config import PlatformConfig # noqa: E402
|
||||
from plugins.platforms.telegram import adapter as tg # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
@@ -1,36 +1,9 @@
|
||||
"""Tests for Telegram model picker thread fallback."""
|
||||
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
mod = MagicMock()
|
||||
mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
mod.constants.ParseMode.MARKDOWN = "Markdown"
|
||||
mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
mod.constants.ParseMode.HTML = "HTML"
|
||||
mod.constants.ChatType.PRIVATE = "private"
|
||||
mod.constants.ChatType.GROUP = "group"
|
||||
mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
mod.constants.ChatType.CHANNEL = "channel"
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter
|
||||
|
||||
|
||||
@@ -9,32 +9,11 @@ rather than silently leaving polling dead.
|
||||
import ast
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import GatewayConfig, Platform, PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram import adapter as tg_adapter # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
from gateway.run import GatewayRunner # noqa: E402
|
||||
|
||||
@@ -12,28 +12,11 @@ entirely (``running=False``) with no reconnect in flight — the long-poll task
|
||||
is gone, so the gateway silently stops receiving messages while the process
|
||||
stays alive (#55769) — and feeds it into the same recovery ladder.
|
||||
"""
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
mod = MagicMock()
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -10,28 +10,6 @@ reconnect is a reliable hung-poll signature.
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import sys
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
telegram_mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
telegram_mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
sys.modules.setdefault("telegram.error", telegram_mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from gateway.config import Platform # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
@@ -6,31 +6,11 @@ Covers the threading behavior control for multi-chunk replies:
|
||||
- "all": All chunks thread to original message
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
from unittest.mock import MagicMock, AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig, GatewayConfig, Platform, _apply_env_overrides, load_gateway_config
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
"""Mock the telegram package if it's not installed."""
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
mod = MagicMock()
|
||||
mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
mod.constants.ChatType.GROUP = "group"
|
||||
mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
mod.constants.ChatType.CHANNEL = "channel"
|
||||
mod.constants.ChatType.PRIVATE = "private"
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -8,31 +8,9 @@ message as ``reply_to_text``, which can cause it to act on unrelated
|
||||
actionable-looking text the user did not quote (#22619).
|
||||
"""
|
||||
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -13,28 +13,11 @@ These tests pin:
|
||||
killing draft streaming for the whole response.
|
||||
3. A non-BadRequest failure propagates so the caller falls back to edit.
|
||||
"""
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
mod = MagicMock()
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
import plugins.platforms.telegram.adapter as tg_mod # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
@@ -5,28 +5,11 @@ can enter a wedged state where ``bot.send_message()`` returns a valid Message
|
||||
but nothing reaches the recipient. ``_send_path_degraded`` short-circuits
|
||||
``send()`` so cron's live-adapter branch falls through to standalone HTTP.
|
||||
"""
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
mod = MagicMock()
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -10,30 +10,6 @@ import pytest
|
||||
_repo = str(Path(__file__).resolve().parents[2])
|
||||
if _repo not in sys.path:
|
||||
sys.path.insert(0, _repo)
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
mod = MagicMock()
|
||||
mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
mod.constants.ParseMode.MARKDOWN = "Markdown"
|
||||
mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
mod.constants.ParseMode.HTML = "HTML"
|
||||
mod.constants.ChatType.PRIVATE = "private"
|
||||
mod.constants.ChatType.GROUP = "group"
|
||||
mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
mod.constants.ChatType.CHANNEL = "channel"
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
@@ -14,31 +14,9 @@ feeds the existing retry ladder. These tests patch the timeout down to keep the
|
||||
suite fast.
|
||||
"""
|
||||
import asyncio
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
telegram_mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
telegram_mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
sys.modules.setdefault("telegram.error", telegram_mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram import adapter as tg_adapter # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
@@ -7,32 +7,11 @@ is enabled, the adapter sets it to "Online" on connect and "Offline" on clean
|
||||
disconnect so users can tell whether the gateway is up.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
telegram_mod = MagicMock()
|
||||
telegram_mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
telegram_mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
telegram_mod.constants.ChatType.GROUP = "group"
|
||||
telegram_mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
telegram_mod.constants.ChatType.CHANNEL = "channel"
|
||||
telegram_mod.constants.ChatType.PRIVATE = "private"
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, telegram_mod)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
|
||||
@@ -2,40 +2,13 @@
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
import pytest
|
||||
|
||||
_repo = str(Path(__file__).resolve().parents[2])
|
||||
if _repo not in sys.path:
|
||||
sys.path.insert(0, _repo)
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
|
||||
mod = MagicMock()
|
||||
mod.ext.ContextTypes.DEFAULT_TYPE = type(None)
|
||||
mod.constants.ParseMode.MARKDOWN = "Markdown"
|
||||
mod.constants.ParseMode.MARKDOWN_V2 = "MarkdownV2"
|
||||
mod.constants.ParseMode.HTML = "HTML"
|
||||
mod.constants.ChatType.PRIVATE = "private"
|
||||
mod.constants.ChatType.GROUP = "group"
|
||||
mod.constants.ChatType.SUPERGROUP = "supergroup"
|
||||
mod.constants.ChatType.CHANNEL = "channel"
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.RetryAfter = type("RetryAfter", (Exception,), {"__init__": lambda self, retry_after=1: setattr(self, "retry_after", retry_after)})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter
|
||||
|
||||
|
||||
@@ -7,28 +7,11 @@ caption to MarkdownV2 (when the formatted text fits the 1024-char caption
|
||||
cap) and falls back to the plain truncated caption when the Bot API rejects
|
||||
the entities or formatting overflows.
|
||||
"""
|
||||
import sys
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
mod = MagicMock()
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram import adapter as telegram_mod # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import TelegramAdapter # noqa: E402
|
||||
|
||||
|
||||
@@ -24,22 +24,6 @@ from unittest.mock import AsyncMock, MagicMock
|
||||
import pytest
|
||||
|
||||
from gateway.config import PlatformConfig
|
||||
|
||||
|
||||
def _ensure_telegram_mock():
|
||||
if "telegram" in sys.modules and hasattr(sys.modules["telegram"], "__file__"):
|
||||
return
|
||||
mod = MagicMock()
|
||||
mod.error.NetworkError = type("NetworkError", (OSError,), {})
|
||||
mod.error.TimedOut = type("TimedOut", (OSError,), {})
|
||||
mod.error.BadRequest = type("BadRequest", (Exception,), {})
|
||||
for name in ("telegram", "telegram.ext", "telegram.constants", "telegram.request"):
|
||||
sys.modules.setdefault(name, mod)
|
||||
sys.modules.setdefault("telegram.error", mod.error)
|
||||
|
||||
|
||||
_ensure_telegram_mock()
|
||||
|
||||
from plugins.platforms.telegram import adapter as telegram_mod # noqa: E402
|
||||
from plugins.platforms.telegram.adapter import ( # noqa: E402
|
||||
TelegramAdapter,
|
||||
|
||||
@@ -182,7 +182,13 @@ class TestGetServicePidsAllProfiles:
|
||||
|
||||
def test_default_scope_uses_current_profile_label(self):
|
||||
"""Without all_profiles, only the current profile's launchd agent is
|
||||
queried (``launchctl list <label>``)."""
|
||||
located (per-label domain-explicit probe, #73627)."""
|
||||
located = []
|
||||
|
||||
def _fake_locate(label):
|
||||
located.append(label)
|
||||
return ("gui/501", 123)
|
||||
|
||||
with (
|
||||
patch("hermes_cli.gateway.is_macos", return_value=True),
|
||||
patch("hermes_cli.gateway.supports_systemd_services", return_value=False),
|
||||
@@ -190,55 +196,81 @@ class TestGetServicePidsAllProfiles:
|
||||
"hermes_cli.gateway.get_launchd_label",
|
||||
return_value="ai.hermes.gateway.myprofile",
|
||||
),
|
||||
patch(
|
||||
"hermes_cli.gateway._locate_launchd_gateway_service",
|
||||
side_effect=_fake_locate,
|
||||
),
|
||||
patch("subprocess.run") as mock_run,
|
||||
):
|
||||
mock_run.return_value = MagicMock(
|
||||
returncode=0, stdout="123\t0\tai.hermes.gateway.myprofile", stderr=""
|
||||
)
|
||||
pids = gateway_mod._get_service_pids()
|
||||
|
||||
assert pids == {123}
|
||||
# Must have used ``launchctl list <label>``, not bare ``launchctl list``
|
||||
# Default scope: exactly the current profile's label, no fleet
|
||||
# enumeration and no bare `launchctl list` scan.
|
||||
assert located == ["ai.hermes.gateway.myprofile"]
|
||||
launchctl_calls = [
|
||||
c[0][0]
|
||||
for c in mock_run.call_args_list
|
||||
if c[0] and c[0][0] and c[0][0][0] == "launchctl"
|
||||
]
|
||||
assert len(launchctl_calls) == 1
|
||||
assert launchctl_calls[0][1] == "list"
|
||||
assert launchctl_calls[0][2] == "ai.hermes.gateway.myprofile"
|
||||
assert launchctl_calls == []
|
||||
|
||||
def test_all_profiles_enumerates_all_gateway_labels(self):
|
||||
"""With all_profiles=True, ``launchctl list`` is called without a label
|
||||
filter, and every row whose last column starts with ``ai.hermes.gateway``
|
||||
is collected."""
|
||||
"""With all_profiles=True, every install-derived gateway label is
|
||||
located (#73627), and the bare ``launchctl list`` prefix scan still
|
||||
widens the EXCLUDE set with unmapped ai.hermes.gateway* agents
|
||||
(#74075 belt-and-suspenders)."""
|
||||
located = []
|
||||
label_pids = {
|
||||
"ai.hermes.gateway": 123,
|
||||
"ai.hermes.gateway-profile-b": 456,
|
||||
}
|
||||
|
||||
def _fake_locate(label):
|
||||
located.append(label)
|
||||
pid = label_pids.get(label)
|
||||
return ("gui/501", pid) if pid else (None, None)
|
||||
|
||||
with (
|
||||
patch("hermes_cli.gateway.is_macos", return_value=True),
|
||||
patch("hermes_cli.gateway.supports_systemd_services", return_value=False),
|
||||
patch(
|
||||
"hermes_cli.gateway.get_launchd_label",
|
||||
return_value="ai.hermes.gateway",
|
||||
),
|
||||
patch(
|
||||
"hermes_cli.gateway.launchd_gateway_labels_for_install",
|
||||
return_value=["ai.hermes.gateway", "ai.hermes.gateway-profile-b"],
|
||||
),
|
||||
patch(
|
||||
"hermes_cli.gateway._locate_launchd_gateway_service",
|
||||
side_effect=_fake_locate,
|
||||
),
|
||||
patch("subprocess.run") as mock_run,
|
||||
):
|
||||
mock_run.return_value = MagicMock(
|
||||
returncode=0,
|
||||
stdout=(
|
||||
"123\t0\tai.hermes.gateway.profile-a\n"
|
||||
"456\t0\tai.hermes.gateway.profile-b\n"
|
||||
"999\t0\tai.hermes.gateway-unmapped\n"
|
||||
"789\t0\tcom.apple.some.other.agent\n"
|
||||
),
|
||||
stderr="",
|
||||
)
|
||||
pids = gateway_mod._get_service_pids(all_profiles=True)
|
||||
|
||||
assert pids == {123, 456}
|
||||
# The non-gateway label (789) must NOT be included.
|
||||
# Label-derived fleet + prefix-scan stragglers; non-gateway excluded.
|
||||
assert pids == {123, 456, 999}
|
||||
assert 789 not in pids
|
||||
# Must have used bare ``launchctl list``
|
||||
assert sorted(located) == [
|
||||
"ai.hermes.gateway",
|
||||
"ai.hermes.gateway-profile-b",
|
||||
]
|
||||
launchctl_calls = [
|
||||
c[0][0]
|
||||
for c in mock_run.call_args_list
|
||||
if c[0] and c[0][0] and c[0][0][0] == "launchctl"
|
||||
]
|
||||
assert len(launchctl_calls) == 1
|
||||
assert launchctl_calls[0] == ["launchctl", "list"]
|
||||
assert launchctl_calls == [["launchctl", "list"]]
|
||||
|
||||
def test_all_profiles_empty_when_no_gateway_labels(self):
|
||||
"""When no ai.hermes.gateway* labels exist, all_profiles returns empty."""
|
||||
|
||||
@@ -78,6 +78,13 @@ class TestFreeRuntime:
|
||||
assert rt is not None
|
||||
assert rt["base_url"] == "https://opencode.ai/zen/v1"
|
||||
|
||||
def test_go_ox_alpha_free_does_not_heal_to_zen(self):
|
||||
"""ox-alpha-free is a KEYED Go-subscription model despite its -free
|
||||
suffix (Zen doesn't serve it; Go 401s anonymous). Membership in the
|
||||
verified keyless catalog — not the suffix — gates the heal."""
|
||||
assert opencode_zen_free_runtime("opencode-go", "ox-alpha-free") is None
|
||||
assert opencode_zen_free_runtime("opencode-zen", "ox-alpha-free") is None
|
||||
|
||||
def test_paid_model_returns_none(self):
|
||||
assert opencode_zen_free_runtime("opencode-zen", "claude-sonnet-5") is None
|
||||
|
||||
@@ -122,3 +129,47 @@ class TestRuntimeProviderKeylessRouting:
|
||||
|
||||
with pytest.raises(AuthError):
|
||||
self._resolve("opencode-zen", "claude-sonnet-5")
|
||||
|
||||
|
||||
class TestKeylessProviderAlwaysAuthenticated:
|
||||
"""opencode-free counts as authenticated everywhere, with zero keys.
|
||||
|
||||
The provider is keyless: there is no credential to configure, so every
|
||||
surface that gates on auth (get_auth_status, provider:model listing,
|
||||
the /model picker source, the desktop explicit-only filter) must treat
|
||||
every install as logged in.
|
||||
"""
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_creds(self, monkeypatch):
|
||||
for var in ("OPENCODE_ZEN_API_KEY", "OPENCODE_GO_API_KEY"):
|
||||
monkeypatch.delenv(var, raising=False)
|
||||
|
||||
def test_auth_status_logged_in(self):
|
||||
from hermes_cli.auth import get_auth_status
|
||||
|
||||
st = get_auth_status("opencode-free")
|
||||
assert st["logged_in"] is True
|
||||
assert st["configured"] is True
|
||||
assert st["key_source"] == "keyless"
|
||||
|
||||
def test_list_available_providers_authenticated(self):
|
||||
from hermes_cli.models import list_available_providers
|
||||
|
||||
rows = {r["id"]: r["authenticated"] for r in list_available_providers()}
|
||||
assert rows.get("opencode-free") is True
|
||||
|
||||
def test_picker_source_includes_provider_with_models(self):
|
||||
import model_tools # noqa: F401 — plugin discovery
|
||||
from hermes_cli.model_switch import list_authenticated_providers
|
||||
|
||||
provs = list_authenticated_providers(for_picker=True)
|
||||
free = [p for p in provs if p["slug"] == "opencode-free"]
|
||||
assert free, "opencode-free must appear in the picker with zero keys"
|
||||
assert free[0]["models"], "picker row must carry the curated models"
|
||||
|
||||
def test_explicit_only_filter_keeps_keyless(self):
|
||||
from hermes_cli.inventory import _provider_is_keyless
|
||||
|
||||
assert _provider_is_keyless("opencode-free") is True
|
||||
assert _provider_is_keyless("opencode-zen") is False
|
||||
|
||||
@@ -294,7 +294,12 @@ class TestTencentTokenhubAgentInit:
|
||||
|
||||
|
||||
class TestTencentTokenhubModelCatalogJSON:
|
||||
"""Verify tencent/hy3:free and tencent/hy3 are present in the website model-catalog.json."""
|
||||
"""Verify tencent/hy3 is present in the website model-catalog.json.
|
||||
|
||||
tencent/hy3:free was delisted 2026-08-21 — the slug vanished from
|
||||
OpenRouter's live catalog (free promo rotated out), so the paid hy3
|
||||
entry is the surviving assertion target.
|
||||
"""
|
||||
|
||||
def test_in_model_catalog_json(self):
|
||||
catalog_path = os.path.join(
|
||||
@@ -318,7 +323,6 @@ class TestTencentTokenhubModelCatalogJSON:
|
||||
for provider_entry in providers:
|
||||
for model in provider_entry.get("models", []):
|
||||
all_ids.add(model.get("id", ""))
|
||||
assert "tencent/hy3:free" in all_ids
|
||||
assert "tencent/hy3" in all_ids
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,171 @@
|
||||
"""Tests for hermes_cli.update_inventory — the plan phase (#91277 Phase 2)."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
import hermes_cli.update_inventory as ui
|
||||
|
||||
|
||||
def _write_state(home: Path, pid: int, sha: str | None = None, version: str | None = None):
|
||||
record = {"pid": pid}
|
||||
if sha:
|
||||
record["code_sha"] = sha
|
||||
if version:
|
||||
record["code_version"] = version
|
||||
(home / "gateway_state.json").write_text(json.dumps(record), encoding="utf-8")
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def fleet(monkeypatch, tmp_path):
|
||||
"""Two profiles with live gateways: default (systemd) + work (manual)."""
|
||||
default_home = tmp_path / "home"
|
||||
work_home = tmp_path / "home" / "profiles" / "work"
|
||||
work_home.mkdir(parents=True)
|
||||
_write_state(default_home, 100, sha="a" * 40, version="1.0")
|
||||
_write_state(work_home, 200) # pre-stamp gateway: no code identity
|
||||
|
||||
import re
|
||||
monkeypatch.setattr("hermes_cli.profiles._get_default_hermes_home", lambda: default_home)
|
||||
monkeypatch.setattr("hermes_cli.profiles._get_profiles_root", lambda: default_home / "profiles")
|
||||
monkeypatch.setattr("hermes_cli.profiles._PROFILE_ID_RE", re.compile(r"^[a-z0-9][a-z0-9_-]*$"), raising=False)
|
||||
monkeypatch.setattr("gateway.status._pid_exists", lambda pid: pid in (100, 200))
|
||||
monkeypatch.setattr("hermes_cli.gateway._get_service_pids", lambda all_profiles=False: {100})
|
||||
monkeypatch.setattr("hermes_cli.gateway.supports_systemd_services", lambda: True)
|
||||
monkeypatch.setattr("hermes_cli.gateway.find_profile_gateway_processes", lambda exclude_pids=None: [])
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.build_info.get_code_identity",
|
||||
lambda refresh=False: {"sha": "a" * 40, "short_sha": "a" * 8, "version": "1.0", "source": "git"},
|
||||
)
|
||||
monkeypatch.setattr("hermes_cli.config.detect_install_method", lambda *a, **k: "git")
|
||||
monkeypatch.setattr("hermes_cli.config.get_managed_system", lambda: None)
|
||||
return tmp_path
|
||||
|
||||
|
||||
class TestCollectInventory:
|
||||
def test_two_profile_fleet(self, fleet):
|
||||
plan = ui.collect_runtime_inventory()
|
||||
assert plan.install_method == "git"
|
||||
assert plan.updatable_in_place is True
|
||||
assert plan.expected_sha == "a" * 40
|
||||
assert plan.profiles == ["default", "work"]
|
||||
assert len(plan.runtimes) == 2
|
||||
by_profile = {r.profile: r for r in plan.runtimes}
|
||||
assert by_profile["default"].pid == 100
|
||||
assert by_profile["default"].supervisor == "systemd"
|
||||
assert by_profile["default"].code_sha == "a" * 40
|
||||
assert by_profile["work"].pid == 200
|
||||
assert by_profile["work"].supervisor == "manual"
|
||||
assert by_profile["work"].code_sha is None # pre-stamp gateway
|
||||
assert "hermes -p work gateway restart" in by_profile["work"].restart_via
|
||||
|
||||
def test_docker_install_not_updatable_in_place(self, fleet, monkeypatch):
|
||||
monkeypatch.setattr("hermes_cli.config.detect_install_method", lambda *a, **k: "docker")
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.config.recommended_update_command_for_method",
|
||||
lambda m: "docker pull nousresearch/hermes-agent:latest",
|
||||
)
|
||||
plan = ui.collect_runtime_inventory()
|
||||
assert plan.install_method == "docker"
|
||||
assert plan.updatable_in_place is False
|
||||
assert "docker pull" in plan.update_mechanism
|
||||
|
||||
def test_dead_pids_excluded(self, fleet, monkeypatch):
|
||||
monkeypatch.setattr("gateway.status._pid_exists", lambda pid: False)
|
||||
plan = ui.collect_runtime_inventory()
|
||||
assert plan.runtimes == []
|
||||
|
||||
def test_pid_file_fallback_covers_unstamped_profiles(self, fleet, monkeypatch):
|
||||
"""Gateways with a PID file but no runtime-status record still appear."""
|
||||
from hermes_cli.gateway import ProfileGatewayProcess
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.gateway.find_profile_gateway_processes",
|
||||
lambda exclude_pids=None: [
|
||||
ProfileGatewayProcess(profile="legacy", path=Path("/x"), pid=300),
|
||||
# duplicate of an already-seen pid — must be deduped
|
||||
ProfileGatewayProcess(profile="default", path=Path("/y"), pid=100),
|
||||
],
|
||||
)
|
||||
monkeypatch.setattr("gateway.status._pid_exists", lambda pid: pid in (100, 200))
|
||||
plan = ui.collect_runtime_inventory()
|
||||
profiles = [r.profile for r in plan.runtimes]
|
||||
assert profiles.count("default") == 1 # deduped by pid
|
||||
assert "legacy" in profiles
|
||||
|
||||
def test_never_raises_when_everything_fails(self, monkeypatch):
|
||||
def _boom(*a, **k):
|
||||
raise RuntimeError("probe down")
|
||||
|
||||
for target in (
|
||||
"hermes_cli.config.detect_install_method",
|
||||
"hermes_cli.build_info.get_code_identity",
|
||||
"hermes_cli.profiles._get_default_hermes_home",
|
||||
"hermes_cli.gateway._get_service_pids",
|
||||
"hermes_cli.gateway.find_profile_gateway_processes",
|
||||
):
|
||||
monkeypatch.setattr(target, _boom)
|
||||
plan = ui.collect_runtime_inventory()
|
||||
assert plan.runtimes == []
|
||||
assert plan.install_method == "unknown"
|
||||
|
||||
def test_plan_serializes_for_receipt(self, fleet):
|
||||
plan = ui.collect_runtime_inventory()
|
||||
payload = plan.to_dict()
|
||||
# must be JSON-clean for the receipt
|
||||
text = json.dumps(payload)
|
||||
restored = json.loads(text)
|
||||
assert restored["install_method"] == "git"
|
||||
assert len(restored["runtimes"]) == 2
|
||||
assert restored["runtimes"][0]["kind"] == "gateway"
|
||||
|
||||
|
||||
class TestPrintPlan:
|
||||
def test_git_fleet_output(self, fleet, capsys):
|
||||
ui.print_update_plan(ui.collect_runtime_inventory())
|
||||
out = capsys.readouterr().out
|
||||
assert "Update plan:" in out
|
||||
assert "Install: git" in out
|
||||
assert "default, work" in out
|
||||
assert "pid 100" in out and "systemd" in out
|
||||
assert "pid 200" in out and "manual" in out
|
||||
|
||||
def test_docker_warns_not_in_place(self, fleet, monkeypatch, capsys):
|
||||
monkeypatch.setattr("hermes_cli.config.detect_install_method", lambda *a, **k: "docker")
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.config.recommended_update_command_for_method",
|
||||
lambda m: "docker pull nousresearch/hermes-agent:latest",
|
||||
)
|
||||
ui.print_update_plan(ui.collect_runtime_inventory())
|
||||
out = capsys.readouterr().out
|
||||
assert "NOT updatable in place" in out
|
||||
assert "docker pull" in out
|
||||
|
||||
def test_empty_fleet_message(self, fleet, monkeypatch, capsys):
|
||||
monkeypatch.setattr("gateway.status._pid_exists", lambda pid: False)
|
||||
ui.print_update_plan(ui.collect_runtime_inventory())
|
||||
assert "none detected" in capsys.readouterr().out
|
||||
|
||||
|
||||
class TestReceiptIntegration:
|
||||
def test_plan_recorded_into_active_receipt(self, fleet, monkeypatch, tmp_path):
|
||||
import hermes_cli.update_receipt as ur
|
||||
|
||||
home = tmp_path / "receipt_home"
|
||||
home.mkdir()
|
||||
monkeypatch.setattr("hermes_cli.config.get_hermes_home", lambda: home, raising=False)
|
||||
ur._current = None
|
||||
ur.begin_update_receipt()
|
||||
plan = ui.collect_runtime_inventory()
|
||||
ui.record_plan_in_receipt(plan)
|
||||
path = ur.finalize_update_receipt("success")
|
||||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||
assert payload["plan"]["install_method"] == "git"
|
||||
assert len(payload["plan"]["runtimes"]) == 2
|
||||
|
||||
def test_noop_without_active_receipt(self, fleet):
|
||||
import hermes_cli.update_receipt as ur
|
||||
|
||||
ur._current = None
|
||||
ui.record_plan_in_receipt(ui.collect_runtime_inventory()) # must not raise
|
||||
@@ -0,0 +1,642 @@
|
||||
"""Regression for #41403 — ``hermes update`` must restart ALL macOS launchd gateways.
|
||||
|
||||
The macOS branch of the update's fleet-restart step only restarted the
|
||||
invoking profile's LaunchAgent (``get_launchd_label()`` is profile-scoped).
|
||||
Sibling ``ai.hermes.gateway-<profile>`` services kept running pre-update
|
||||
modules cached in ``sys.modules`` and died on their next agent turn once the
|
||||
new code lazily imported a symbol the old module generation didn't have
|
||||
(``ImportError: cannot import name ...`` — or, with a wider version gap,
|
||||
``TypeError``/``AttributeError`` on changed call signatures with garbled
|
||||
tracebacks, because the source files on disk no longer match the loaded
|
||||
code objects).
|
||||
|
||||
Also covers the launchd-domain review feedback on PR #41403: every sibling
|
||||
interaction (liveness discovery, kickstart, fresh-PID verification) must be
|
||||
domain-explicit — ``_launchd_domain()`` caches the *current* profile's
|
||||
domain, and a sibling bootstrapped in the other supported domain
|
||||
(``gui/<uid>`` vs ``user/<uid>``) would otherwise be probed or kickstarted
|
||||
in a domain it does not live in.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
import hermes_cli.gateway as gw
|
||||
import hermes_cli.profiles
|
||||
from hermes_cli.gateway import (
|
||||
_locate_launchd_gateway_service,
|
||||
_parse_launchd_pid_from_print_output,
|
||||
_probe_launchd_domain_for_label,
|
||||
launchd_gateway_labels_for_install,
|
||||
)
|
||||
from hermes_cli.update_cmd import (
|
||||
_restart_macos_launchd_gateways,
|
||||
_warn_incomplete_gateway_fleet_restart,
|
||||
)
|
||||
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
sys.platform == "win32",
|
||||
reason="launchd fleet restart is macOS-only; helpers use POSIX os.getuid",
|
||||
)
|
||||
|
||||
UID = 501
|
||||
|
||||
PRINT_RUNNING = (
|
||||
"system/com.example = {\n"
|
||||
"\tactive count = 1\n"
|
||||
"\tstate = running\n"
|
||||
"\tpid = 4242\n"
|
||||
"\tprogram = /usr/bin/true\n"
|
||||
"}\n"
|
||||
)
|
||||
PRINT_LOADED_NOT_RUNNING = (
|
||||
"system/com.example = {\n"
|
||||
"\tactive count = 0\n"
|
||||
"\tstate = not running\n"
|
||||
"\tprogram = /usr/bin/true\n"
|
||||
"}\n"
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _fixed_uid(monkeypatch):
|
||||
monkeypatch.setattr(gw.os, "getuid", lambda: UID)
|
||||
|
||||
|
||||
def _completed(returncode: int = 0, stdout: str = "") -> subprocess.CompletedProcess:
|
||||
return subprocess.CompletedProcess(
|
||||
args=[], returncode=returncode, stdout=stdout, stderr=""
|
||||
)
|
||||
|
||||
|
||||
class _Profile:
|
||||
def __init__(self, name, is_default=False):
|
||||
self.name = name
|
||||
self.is_default = is_default
|
||||
|
||||
|
||||
class TestLaunchdGatewayLabelsForInstall:
|
||||
def test_labels_derive_from_this_installs_profiles(self, monkeypatch):
|
||||
"""The fleet is THIS install's profiles, root first — never a glob of
|
||||
the shared per-user LaunchAgents dir. A sandboxed HERMES_HOME (tests,
|
||||
side-by-side installs) must not enumerate — and restart — another
|
||||
install's services, and the hermetic test suite must not see the dev
|
||||
machine's real fleet."""
|
||||
monkeypatch.setattr(
|
||||
hermes_cli.profiles,
|
||||
"list_profiles",
|
||||
lambda: [
|
||||
_Profile("tfl-wiki"),
|
||||
_Profile("default", is_default=True),
|
||||
_Profile("merit-ops"),
|
||||
_Profile("Bad Name!"), # cannot map to a service suffix — skipped
|
||||
],
|
||||
)
|
||||
assert launchd_gateway_labels_for_install() == [
|
||||
"ai.hermes.gateway",
|
||||
"ai.hermes.gateway-merit-ops",
|
||||
"ai.hermes.gateway-tfl-wiki",
|
||||
]
|
||||
|
||||
def test_no_profiles_means_no_fleet(self, monkeypatch):
|
||||
monkeypatch.setattr(hermes_cli.profiles, "list_profiles", lambda: [])
|
||||
assert launchd_gateway_labels_for_install() == []
|
||||
|
||||
|
||||
class TestParseLaunchdPidFromPrintOutput:
|
||||
def test_running_service_pid(self):
|
||||
assert _parse_launchd_pid_from_print_output(PRINT_RUNNING) == 4242
|
||||
|
||||
def test_loaded_but_not_running_has_no_pid(self):
|
||||
assert _parse_launchd_pid_from_print_output(PRINT_LOADED_NOT_RUNNING) is None
|
||||
|
||||
|
||||
class TestLocateLaunchdGatewayService:
|
||||
def test_domains_resolve_per_label_not_from_cache(self, monkeypatch):
|
||||
"""The #41403 review defect: sibling domains are independent."""
|
||||
gui_loaded = {"ai.hermes.gateway-a"}
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
assert cmd[:2] == ["launchctl", "print"]
|
||||
domain, _, label = cmd[2].rpartition("/")
|
||||
in_gui = domain == f"gui/{UID}" and label in gui_loaded
|
||||
in_user = domain == f"user/{UID}" and label not in gui_loaded
|
||||
if in_gui or in_user:
|
||||
return _completed(0, PRINT_RUNNING)
|
||||
return _completed(113)
|
||||
|
||||
monkeypatch.setattr(gw.subprocess, "run", fake_run)
|
||||
# Simulate a prior current-profile resolution having populated the
|
||||
# process-wide cache — per-label lookups must not consult it.
|
||||
monkeypatch.setattr(gw, "_resolved_launchd_domain", f"gui/{UID}")
|
||||
|
||||
assert _locate_launchd_gateway_service("ai.hermes.gateway-a") == (
|
||||
f"gui/{UID}",
|
||||
4242,
|
||||
)
|
||||
assert _locate_launchd_gateway_service("ai.hermes.gateway-b") == (
|
||||
f"user/{UID}",
|
||||
4242,
|
||||
)
|
||||
|
||||
def test_loaded_without_live_process(self, monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
gw.subprocess,
|
||||
"run",
|
||||
lambda *a, **k: _completed(0, PRINT_LOADED_NOT_RUNNING),
|
||||
)
|
||||
assert _locate_launchd_gateway_service("ai.hermes.gateway-x") == (
|
||||
f"gui/{UID}",
|
||||
None,
|
||||
)
|
||||
|
||||
def test_not_loaded_in_either_domain(self, monkeypatch):
|
||||
monkeypatch.setattr(gw.subprocess, "run", lambda *a, **k: _completed(113))
|
||||
assert _locate_launchd_gateway_service("ai.hermes.gateway-x") == (None, None)
|
||||
|
||||
def test_timeout_propagates_to_caller(self, monkeypatch):
|
||||
"""A wedged launchctl must surface as a failure, not read as
|
||||
'unloaded' — the update path owns per-label failure accounting."""
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
raise subprocess.TimeoutExpired(cmd=cmd, timeout=5)
|
||||
|
||||
monkeypatch.setattr(gw.subprocess, "run", fake_run)
|
||||
with pytest.raises(subprocess.TimeoutExpired):
|
||||
_locate_launchd_gateway_service("ai.hermes.gateway-x")
|
||||
|
||||
|
||||
class TestProbeLaunchdDomainForLabel:
|
||||
def test_unloaded_label_falls_back_to_managername(self, monkeypatch):
|
||||
def fake_run(cmd, **kwargs):
|
||||
if cmd[:2] == ["launchctl", "print"]:
|
||||
raise subprocess.CalledProcessError(113, cmd)
|
||||
if cmd == ["launchctl", "managername"]:
|
||||
return _completed(0, "Aqua\n")
|
||||
raise AssertionError(f"unexpected command {cmd}")
|
||||
|
||||
monkeypatch.setattr(gw.subprocess, "run", fake_run)
|
||||
assert _probe_launchd_domain_for_label("ai.hermes.gateway-x") == f"gui/{UID}"
|
||||
|
||||
def test_unloaded_label_defaults_to_user_domain(self, monkeypatch):
|
||||
def fake_run(cmd, **kwargs):
|
||||
if cmd[:2] == ["launchctl", "print"]:
|
||||
raise subprocess.CalledProcessError(113, cmd)
|
||||
if cmd == ["launchctl", "managername"]:
|
||||
return _completed(0, "Background\n")
|
||||
raise AssertionError(f"unexpected command {cmd}")
|
||||
|
||||
monkeypatch.setattr(gw.subprocess, "run", fake_run)
|
||||
assert _probe_launchd_domain_for_label("ai.hermes.gateway-x") == f"user/{UID}"
|
||||
|
||||
|
||||
class TestGetServicePidsScoping:
|
||||
def _wire(self, monkeypatch):
|
||||
monkeypatch.setattr(gw, "is_macos", lambda: True)
|
||||
monkeypatch.setattr(gw, "supports_systemd_services", lambda: False)
|
||||
monkeypatch.setattr(gw, "get_launchd_label", lambda: "ai.hermes.gateway")
|
||||
monkeypatch.setattr(
|
||||
gw,
|
||||
"launchd_gateway_labels_for_install",
|
||||
lambda: ["ai.hermes.gateway", "ai.hermes.gateway-a", "ai.hermes.gateway-b"],
|
||||
)
|
||||
located = {
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
"ai.hermes.gateway-a": (f"gui/{UID}", 200),
|
||||
"ai.hermes.gateway-b": (None, None), # not bootstrapped
|
||||
}
|
||||
monkeypatch.setattr(
|
||||
gw, "_locate_launchd_gateway_service", lambda label: located[label]
|
||||
)
|
||||
|
||||
def test_all_profiles_returns_every_gateway_service_pid(self, monkeypatch):
|
||||
"""The update sweep's exclude-set must protect ALL freshly-restarted
|
||||
services, not only the invoking profile's (else the sweep SIGTERMs
|
||||
gateways launchd just respawned)."""
|
||||
self._wire(monkeypatch)
|
||||
assert gw._get_service_pids(all_profiles=True) == {100, 200}
|
||||
|
||||
def test_default_stays_scoped_to_current_profile(self, monkeypatch):
|
||||
"""Regression guard: default-scope callers (gateway status, cron,
|
||||
stop_profile_gateway's orphan reaper) must NOT start seeing sibling
|
||||
service PIDs — the reaper SIGTERM/SIGKILLs what they feed it."""
|
||||
self._wire(monkeypatch)
|
||||
assert gw._get_service_pids() == {100}
|
||||
|
||||
def test_find_gateway_pids_passes_profile_scope_through(self, monkeypatch):
|
||||
calls: list[bool] = []
|
||||
monkeypatch.setattr(
|
||||
gw,
|
||||
"_get_service_pids",
|
||||
lambda all_profiles=False: (calls.append(all_profiles), set())[1],
|
||||
)
|
||||
monkeypatch.setattr(gw, "_scan_gateway_pids", lambda *a, **k: [])
|
||||
monkeypatch.setattr(gw, "supports_systemd_services", lambda: True)
|
||||
|
||||
gw.find_gateway_pids(all_profiles=False)
|
||||
gw.find_gateway_pids(all_profiles=True)
|
||||
assert calls == [False, True]
|
||||
|
||||
|
||||
def _fleet(monkeypatch, tmp_path, *, current, labels, located,
|
||||
registered=None, plist_exists=True,
|
||||
drain_results=None, kick_errors=None, wait_results=None):
|
||||
"""Wire a fake launchd fleet through hermes_cli.gateway seams.
|
||||
|
||||
``located`` maps label -> (domain, pid) as ``_locate_launchd_gateway_service``
|
||||
would return it (values may also be exceptions to raise). ``registered``
|
||||
maps label -> bool for the current-profile ``launchctl list`` gate and
|
||||
defaults to "located in some domain". Returns a SimpleNamespace of
|
||||
recorder lists: rec.kickstarts, rec.drains, rec.current_restarts, rec.waits, locates,
|
||||
registered_checks.
|
||||
"""
|
||||
from types import SimpleNamespace
|
||||
|
||||
rec = SimpleNamespace(
|
||||
kickstarts=[], drains=[], current_restarts=[], waits=[],
|
||||
locates=[], registered_checks=[],
|
||||
)
|
||||
|
||||
plist = tmp_path / f"{current}.plist"
|
||||
if plist_exists:
|
||||
plist.write_text("<plist/>")
|
||||
|
||||
def fake_locate(label):
|
||||
rec.locates.append(label)
|
||||
value = located[label]
|
||||
if isinstance(value, Exception):
|
||||
raise value
|
||||
return value
|
||||
|
||||
def fake_registered(label):
|
||||
rec.registered_checks.append(label)
|
||||
if registered is not None:
|
||||
return registered[label]
|
||||
value = located.get(label)
|
||||
return (
|
||||
value is not None
|
||||
and not isinstance(value, Exception)
|
||||
and value[0] is not None
|
||||
)
|
||||
|
||||
monkeypatch.setattr(gw, "get_launchd_label", lambda: current)
|
||||
monkeypatch.setattr(gw, "get_launchd_plist_path", lambda: plist)
|
||||
monkeypatch.setattr(gw, "launchd_gateway_labels_for_install", lambda: list(labels))
|
||||
monkeypatch.setattr(gw, "_locate_launchd_gateway_service", fake_locate)
|
||||
monkeypatch.setattr(gw, "_launchd_service_registered", fake_registered)
|
||||
monkeypatch.setattr(
|
||||
gw,
|
||||
"_graceful_restart_via_sigusr1",
|
||||
lambda pid, drain_timeout: (rec.drains.append(pid), (drain_results or {}).get(pid, False))[1],
|
||||
)
|
||||
|
||||
def fake_kickstart(label, domain):
|
||||
err = (kick_errors or {}).get(label)
|
||||
if err is not None:
|
||||
raise err
|
||||
rec.kickstarts.append(f"{domain}/{label}")
|
||||
|
||||
monkeypatch.setattr(gw, "_launchd_kickstart", fake_kickstart)
|
||||
|
||||
def fake_wait(label, old_pid, timeout, domain):
|
||||
rec.waits.append(f"{domain}/{label}")
|
||||
return (wait_results or {}).get(label, True)
|
||||
|
||||
monkeypatch.setattr(gw, "_wait_for_launchd_service_pid", fake_wait)
|
||||
monkeypatch.setattr(
|
||||
gw, "launchd_restart", lambda: rec.current_restarts.append(current)
|
||||
)
|
||||
return rec
|
||||
|
||||
|
||||
class TestRestartMacosLaunchdGateways:
|
||||
def test_current_delegates_and_siblings_kickstart_in_own_domains(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
"""Current profile keeps launchd_restart(); every sibling (including
|
||||
the root gateway when a named profile invokes the update) is
|
||||
kickstarted — and verified — in the domain IT was located in."""
|
||||
current = "ai.hermes.gateway-merit-ops"
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current=current,
|
||||
labels=["ai.hermes.gateway", current, "ai.hermes.gateway-user-scoped"],
|
||||
located={
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
current: (f"gui/{UID}", 200),
|
||||
"ai.hermes.gateway-user-scoped": (f"user/{UID}", 300),
|
||||
},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert rec.current_restarts == [current]
|
||||
assert rec.kickstarts == [
|
||||
f"gui/{UID}/ai.hermes.gateway",
|
||||
f"user/{UID}/ai.hermes.gateway-user-scoped",
|
||||
]
|
||||
assert rec.waits == [
|
||||
f"gui/{UID}/ai.hermes.gateway",
|
||||
f"user/{UID}/ai.hermes.gateway-user-scoped",
|
||||
]
|
||||
assert restarted == [
|
||||
current,
|
||||
"ai.hermes.gateway",
|
||||
"ai.hermes.gateway-user-scoped",
|
||||
]
|
||||
assert failed == []
|
||||
# Siblings were drained before the hard kickstart.
|
||||
assert set(rec.drains) == {100, 300}
|
||||
|
||||
def test_current_profile_without_plist_makes_no_launchctl_calls(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
"""Upstream gate order preserved: no plist → the current profile is
|
||||
skipped without ANY launchctl interaction (no registered probe, no
|
||||
locate) — and definitely without inventing a failure. Siblings are
|
||||
still processed."""
|
||||
current = "ai.hermes.gateway"
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current=current,
|
||||
labels=[current, "ai.hermes.gateway-a"],
|
||||
located={"ai.hermes.gateway-a": (f"gui/{UID}", 200)},
|
||||
plist_exists=False,
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert rec.current_restarts == []
|
||||
assert current not in rec.registered_checks
|
||||
assert current not in rec.locates
|
||||
assert restarted == ["ai.hermes.gateway-a"]
|
||||
assert failed == []
|
||||
|
||||
def test_current_profile_registered_but_unlocatable_still_restarts(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
"""macOS-26 quirk: a label can be `launchctl list`-registered while
|
||||
both explicit gui/user `launchctl print` probes fail (domain doesn't
|
||||
support service management). The gate must use the registered
|
||||
predicate and hand off to launchd_restart(), which owns the
|
||||
domain-unsupported fallback — locate is for siblings only."""
|
||||
current = "ai.hermes.gateway"
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current=current,
|
||||
labels=[current],
|
||||
located={current: (None, None)},
|
||||
registered={current: True},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert rec.current_restarts == [current]
|
||||
assert current not in rec.locates
|
||||
assert restarted == [current]
|
||||
assert failed == []
|
||||
|
||||
def test_unbootstrapped_sibling_is_skipped_not_failed(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current="ai.hermes.gateway",
|
||||
labels=["ai.hermes.gateway", "ai.hermes.gateway-idle"],
|
||||
located={
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
"ai.hermes.gateway-idle": (None, None),
|
||||
},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert rec.kickstarts == []
|
||||
assert restarted == ["ai.hermes.gateway"]
|
||||
assert failed == []
|
||||
|
||||
def test_loaded_but_not_running_sibling_is_kickstarted(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
"""A bootstrapped service with no live process still holds the old
|
||||
code path for its next launch trigger — kickstart it (no drain)."""
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current="ai.hermes.gateway",
|
||||
labels=["ai.hermes.gateway", "ai.hermes.gateway-dormant"],
|
||||
located={
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
"ai.hermes.gateway-dormant": (f"gui/{UID}", None),
|
||||
},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert rec.drains == []
|
||||
assert rec.kickstarts == [f"gui/{UID}/ai.hermes.gateway-dormant"]
|
||||
assert restarted == ["ai.hermes.gateway", "ai.hermes.gateway-dormant"]
|
||||
assert failed == []
|
||||
|
||||
def test_graceful_drain_with_keepalive_respawn_skips_kickstart(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
"""When SIGUSR1 rec.drains the sibling and KeepAlive already respawned it
|
||||
on a fresh PID, a second hard kickstart would kill the new process."""
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current="ai.hermes.gateway",
|
||||
labels=["ai.hermes.gateway", "ai.hermes.gateway-a"],
|
||||
located={
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
"ai.hermes.gateway-a": (f"gui/{UID}", 200),
|
||||
},
|
||||
drain_results={200: True},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert rec.drains == [200]
|
||||
assert rec.kickstarts == []
|
||||
assert rec.waits == [f"gui/{UID}/ai.hermes.gateway-a"]
|
||||
assert restarted == ["ai.hermes.gateway", "ai.hermes.gateway-a"]
|
||||
assert failed == []
|
||||
|
||||
def test_kickstart_failure_is_recorded_and_rest_continue(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current="ai.hermes.gateway",
|
||||
labels=[
|
||||
"ai.hermes.gateway",
|
||||
"ai.hermes.gateway-bad",
|
||||
"ai.hermes.gateway-good",
|
||||
],
|
||||
located={
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
"ai.hermes.gateway-bad": (f"gui/{UID}", 200),
|
||||
"ai.hermes.gateway-good": (f"gui/{UID}", 300),
|
||||
},
|
||||
kick_errors={
|
||||
"ai.hermes.gateway-bad": subprocess.CalledProcessError(
|
||||
5, ["launchctl", "kickstart"]
|
||||
)
|
||||
},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert failed == ["ai.hermes.gateway-bad"]
|
||||
assert rec.kickstarts == [f"gui/{UID}/ai.hermes.gateway-good"]
|
||||
assert restarted == ["ai.hermes.gateway", "ai.hermes.gateway-good"]
|
||||
|
||||
def test_timeout_during_discovery_is_failed_and_rest_continue(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
"""A wedged launchctl during liveness discovery must be accounted as
|
||||
a failure (the sibling may still be on old code), not silently
|
||||
skipped — and must not abort the remaining fleet (#68523 parity)."""
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current="ai.hermes.gateway",
|
||||
labels=[
|
||||
"ai.hermes.gateway",
|
||||
"ai.hermes.gateway-wedged",
|
||||
"ai.hermes.gateway-after",
|
||||
],
|
||||
located={
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
"ai.hermes.gateway-wedged": subprocess.TimeoutExpired(
|
||||
cmd=["launchctl", "print"], timeout=5
|
||||
),
|
||||
"ai.hermes.gateway-after": (f"gui/{UID}", 300),
|
||||
},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert failed == ["ai.hermes.gateway-wedged"]
|
||||
assert rec.kickstarts == [f"gui/{UID}/ai.hermes.gateway-after"]
|
||||
assert restarted == ["ai.hermes.gateway", "ai.hermes.gateway-after"]
|
||||
|
||||
def test_timeout_during_kickstart_is_failed_and_rest_continue(
|
||||
self, monkeypatch, tmp_path
|
||||
):
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current="ai.hermes.gateway",
|
||||
labels=[
|
||||
"ai.hermes.gateway",
|
||||
"ai.hermes.gateway-wedged",
|
||||
"ai.hermes.gateway-after",
|
||||
],
|
||||
located={
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
"ai.hermes.gateway-wedged": (f"gui/{UID}", 200),
|
||||
"ai.hermes.gateway-after": (f"gui/{UID}", 300),
|
||||
},
|
||||
kick_errors={
|
||||
"ai.hermes.gateway-wedged": subprocess.TimeoutExpired(
|
||||
cmd=["launchctl", "kickstart"], timeout=90
|
||||
)
|
||||
},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert failed == ["ai.hermes.gateway-wedged"]
|
||||
assert rec.kickstarts == [f"gui/{UID}/ai.hermes.gateway-after"]
|
||||
assert restarted == ["ai.hermes.gateway", "ai.hermes.gateway-after"]
|
||||
|
||||
def test_sibling_that_never_comes_back_is_failed(self, monkeypatch, tmp_path):
|
||||
rec = _fleet(
|
||||
monkeypatch,
|
||||
tmp_path,
|
||||
current="ai.hermes.gateway",
|
||||
labels=["ai.hermes.gateway", "ai.hermes.gateway-zombie"],
|
||||
located={
|
||||
"ai.hermes.gateway": (f"gui/{UID}", 100),
|
||||
"ai.hermes.gateway-zombie": (f"gui/{UID}", 200),
|
||||
},
|
||||
wait_results={"ai.hermes.gateway-zombie": False},
|
||||
)
|
||||
restarted: list[str] = []
|
||||
failed: list[str] = []
|
||||
|
||||
_restart_macos_launchd_gateways(restarted, failed, drain_budget=0.0)
|
||||
|
||||
assert restarted == ["ai.hermes.gateway"]
|
||||
assert failed == ["ai.hermes.gateway-zombie"]
|
||||
|
||||
|
||||
class TestWaitForLaunchdServicePid:
|
||||
def test_returns_true_once_pid_changes(self, monkeypatch):
|
||||
pids = iter([200, 200, 4242])
|
||||
monkeypatch.setattr(
|
||||
gw,
|
||||
"_launchd_print_service_pid",
|
||||
lambda domain, label: (True, next(pids)),
|
||||
)
|
||||
monkeypatch.setattr(gw.time, "sleep", lambda _s: None)
|
||||
assert gw._wait_for_launchd_service_pid(
|
||||
"ai.hermes.gateway-x", old_pid=200, timeout=5.0, domain=f"gui/{UID}"
|
||||
)
|
||||
|
||||
def test_returns_false_when_pid_never_changes(self, monkeypatch):
|
||||
clock = iter(float(i) for i in range(100))
|
||||
monkeypatch.setattr(gw.time, "monotonic", lambda: next(clock))
|
||||
monkeypatch.setattr(gw.time, "sleep", lambda _s: None)
|
||||
monkeypatch.setattr(
|
||||
gw,
|
||||
"_launchd_print_service_pid",
|
||||
lambda domain, label: (True, 200),
|
||||
)
|
||||
assert not gw._wait_for_launchd_service_pid(
|
||||
"ai.hermes.gateway-x", old_pid=200, timeout=3.0, domain=f"gui/{UID}"
|
||||
)
|
||||
|
||||
|
||||
class TestIncompleteWarningMentionsLaunchctl:
|
||||
def test_launchd_labels_get_launchctl_hint(self, capsys):
|
||||
_warn_incomplete_gateway_fleet_restart(["ai.hermes.gateway-merit-ops"])
|
||||
out = capsys.readouterr().out
|
||||
assert "Update incomplete" in out
|
||||
assert "launchctl kickstart -k" in out
|
||||
|
||||
def test_systemd_units_keep_systemctl_hint(self, capsys):
|
||||
_warn_incomplete_gateway_fleet_restart(["hermes-gateway-coder"])
|
||||
out = capsys.readouterr().out
|
||||
assert "systemctl" in out
|
||||
assert "launchctl" not in out
|
||||
@@ -11,6 +11,7 @@ Covers:
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
@@ -121,6 +122,111 @@ class TestReceiptLifecycle:
|
||||
assert ur.read_latest_receipt() is None
|
||||
|
||||
|
||||
class TestCommandBoundaryFinalization:
|
||||
"""Receipt lifetime is owned by the update-command boundary (#91283 review).
|
||||
|
||||
Early sys.exit paths (concurrent-instance preflight exit-2, venv-holder
|
||||
refusal, fetch failure) predate the inner finalize sites; the boundary
|
||||
safety net must persist the receipt exactly once with the stop reason,
|
||||
while inner-finalized runs are untouched.
|
||||
"""
|
||||
|
||||
def test_pending_receipt_persisted_on_exit_2_refusal(self, receipt_home):
|
||||
ur.begin_update_receipt()
|
||||
ur.record_step("windows_preflight", False, "another hermes.exe running")
|
||||
path = ur.finalize_pending_update_receipt(2, "sys.exit(2)")
|
||||
assert path is not None and path.is_file()
|
||||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||
assert payload["outcome"] == "refused"
|
||||
assert payload["exit_code"] == 2
|
||||
assert payload["stop_reason"] == "sys.exit(2)"
|
||||
assert payload["finished_at"] is not None
|
||||
assert ur._current is None
|
||||
|
||||
def test_pending_receipt_persisted_on_exit_1_failure(self, receipt_home):
|
||||
ur.begin_update_receipt()
|
||||
path = ur.finalize_pending_update_receipt(1, "sys.exit(1)")
|
||||
payload = json.loads(path.read_text(encoding="utf-8"))
|
||||
assert payload["outcome"] == "failed"
|
||||
assert payload["exit_code"] == 1
|
||||
|
||||
def test_noop_when_inner_path_already_finalized(self, receipt_home):
|
||||
"""Exactly-once: boundary call after an inner finalize writes nothing."""
|
||||
ur.begin_update_receipt()
|
||||
first = ur.finalize_update_receipt("success")
|
||||
assert first is not None
|
||||
second = ur.finalize_pending_update_receipt(0, "boundary")
|
||||
assert second is None
|
||||
directory = receipt_home / "logs" / "update_receipts"
|
||||
assert len(list(directory.glob("update_*.json"))) == 1
|
||||
|
||||
def test_noop_when_never_begun(self, receipt_home):
|
||||
assert ur.finalize_pending_update_receipt(2, "sys.exit(2)") is None
|
||||
assert ur.read_latest_receipt() is None
|
||||
|
||||
def test_cmd_update_boundary_finalizes_on_early_exit(
|
||||
self, receipt_home, monkeypatch
|
||||
):
|
||||
"""End-to-end through the real cmd_update wrapper: an impl that begins
|
||||
a receipt then sys.exit(2)s (the concurrent-instance shape) must leave
|
||||
a finalized 'refused' receipt, preserve the exit code, and clear the
|
||||
singleton."""
|
||||
from types import SimpleNamespace
|
||||
|
||||
from hermes_cli import main as hermes_main
|
||||
|
||||
def _fake_impl(args, gateway_mode):
|
||||
ur.begin_update_receipt()
|
||||
ur.record_step("windows_preflight", False, "hermes.exe holds venv")
|
||||
sys.exit(2)
|
||||
|
||||
monkeypatch.setattr(hermes_main, "_cmd_update_impl", _fake_impl)
|
||||
monkeypatch.setattr(
|
||||
hermes_main, "detect_install_method", lambda *a, **k: "git", raising=False
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
hermes_main,
|
||||
"_install_hangup_protection",
|
||||
lambda gateway_mode: None,
|
||||
raising=False,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
hermes_main, "_finalize_update_output", lambda state: None, raising=False
|
||||
)
|
||||
|
||||
class _FakeLock:
|
||||
holder = None
|
||||
|
||||
def acquire(self):
|
||||
return True
|
||||
|
||||
def release(self):
|
||||
pass
|
||||
|
||||
import hermes_cli.update_lock as update_lock_mod
|
||||
|
||||
monkeypatch.setattr(update_lock_mod, "UpdateLock", _FakeLock)
|
||||
|
||||
args = SimpleNamespace(
|
||||
check=False, gateway=False, branch=None, yes=False,
|
||||
force=False, force_venv=False,
|
||||
)
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
hermes_main.cmd_update(args)
|
||||
|
||||
assert exc_info.value.code == 2 # exit code preserved
|
||||
latest = ur.read_latest_receipt()
|
||||
assert latest is not None
|
||||
assert latest["outcome"] == "refused"
|
||||
assert latest["exit_code"] == 2
|
||||
assert latest["stop_reason"] == "sys.exit(2)"
|
||||
assert latest["steps"][0]["name"] == "windows_preflight"
|
||||
assert ur._current is None
|
||||
# exactly-once: exactly one receipt file
|
||||
directory = receipt_home / "logs" / "update_receipts"
|
||||
assert len(list(directory.glob("update_*.json"))) == 1
|
||||
|
||||
|
||||
class TestFleetClassification:
|
||||
def _fleet_with(self, monkeypatch, tmp_path, record, expected_sha="a" * 40):
|
||||
"""Run collect_fleet_versions against one fake default profile."""
|
||||
|
||||
@@ -16,6 +16,96 @@ def opencode_go_profile():
|
||||
return profile
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def opencode_zen_profile():
|
||||
"""Resolve the registered OpenCode Zen provider profile."""
|
||||
import model_tools # noqa: F401
|
||||
import providers
|
||||
|
||||
profile = providers.get_provider_profile("opencode-zen")
|
||||
assert profile is not None, "opencode-zen provider profile must be registered"
|
||||
return profile
|
||||
|
||||
|
||||
class TestOpenCodeZenOxReasoning:
|
||||
"""Ox Alpha Free uses OpenCode Zen's native reasoning_effort control."""
|
||||
|
||||
def test_max_effort_is_emitted(self, opencode_zen_profile):
|
||||
extra_body, top_level = opencode_zen_profile.build_api_kwargs_extras(
|
||||
reasoning_config={"enabled": True, "effort": "max"},
|
||||
model="x-preview-f-free",
|
||||
)
|
||||
assert extra_body == {}
|
||||
assert top_level == {"reasoning_effort": "max"}
|
||||
|
||||
@pytest.mark.parametrize("reasoning_config", [None, {"enabled": False}])
|
||||
def test_unset_or_disabled_preserves_server_default(
|
||||
self, opencode_zen_profile, reasoning_config
|
||||
):
|
||||
extra_body, top_level = opencode_zen_profile.build_api_kwargs_extras(
|
||||
reasoning_config=reasoning_config,
|
||||
model="x-preview-f-free",
|
||||
)
|
||||
assert extra_body == {}
|
||||
assert top_level == {}
|
||||
|
||||
def test_other_zen_models_are_untouched(self, opencode_zen_profile):
|
||||
extra_body, top_level = opencode_zen_profile.build_api_kwargs_extras(
|
||||
reasoning_config={"enabled": True, "effort": "max"},
|
||||
model="gemini-3-flash",
|
||||
)
|
||||
assert extra_body == {}
|
||||
assert top_level == {}
|
||||
|
||||
def test_max_reaches_chat_completions_request(self, opencode_zen_profile):
|
||||
from agent.transports.chat_completions import ChatCompletionsTransport
|
||||
|
||||
kwargs = ChatCompletionsTransport().build_kwargs(
|
||||
model="x-preview-f-free",
|
||||
messages=[{"role": "user", "content": "ping"}],
|
||||
tools=None,
|
||||
provider_profile=opencode_zen_profile,
|
||||
reasoning_config={"enabled": True, "effort": "max"},
|
||||
base_url="https://opencode.ai/zen/v1",
|
||||
)
|
||||
assert "extra_body" not in kwargs
|
||||
assert kwargs["reasoning_effort"] == "max"
|
||||
|
||||
def test_unsupported_efforts_clamp_to_wire_vocabulary(self, opencode_zen_profile):
|
||||
"""medium/xhigh are not on Ox Alpha's wire (400 raw); they must clamp
|
||||
to the nearest supported level, never pass through."""
|
||||
for requested, expected in (("medium", "low"), ("xhigh", "max")):
|
||||
_, top_level = opencode_zen_profile.build_api_kwargs_extras(
|
||||
reasoning_config={"enabled": True, "effort": requested},
|
||||
model="x-preview-f-free",
|
||||
)
|
||||
assert top_level == {"reasoning_effort": expected}, requested
|
||||
|
||||
def test_opencode_free_profile_shares_the_translation(self):
|
||||
"""Ox Alpha is reachable via the keyless opencode-free provider too;
|
||||
its profile must emit the identical clamped reasoning_effort."""
|
||||
import model_tools # noqa: F401
|
||||
import providers
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
profile = providers.get_provider_profile("opencode-free")
|
||||
assert profile is not None
|
||||
assert (
|
||||
type(profile).build_api_kwargs_extras
|
||||
is not ProviderProfile.build_api_kwargs_extras
|
||||
), "opencode-free must override build_api_kwargs_extras (aux gate)"
|
||||
_, top_level = profile.build_api_kwargs_extras(
|
||||
reasoning_config={"enabled": True, "effort": "medium"},
|
||||
model="x-preview-f-free",
|
||||
)
|
||||
assert top_level == {"reasoning_effort": "low"}
|
||||
_, other = profile.build_api_kwargs_extras(
|
||||
reasoning_config={"enabled": True, "effort": "max"},
|
||||
model="big-pickle",
|
||||
)
|
||||
assert other == {}
|
||||
|
||||
|
||||
class TestOpenCodeGoKimiReasoning:
|
||||
"""Kimi K2 models use Moonshot's thinking + reasoning_effort shape on OpenCode Go."""
|
||||
|
||||
|
||||
@@ -2,10 +2,66 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
|
||||
import run_agent as run_agent_module
|
||||
from run_agent import AIAgent
|
||||
|
||||
|
||||
_REAL_THREAD = threading.Thread
|
||||
|
||||
|
||||
class _TurnBoundaryReached(Exception):
|
||||
"""Stop a live turn exactly when it reaches turn-context construction."""
|
||||
|
||||
|
||||
class CapturingThread:
|
||||
targets = []
|
||||
|
||||
def __init__(self, *, target, daemon=None, name=None):
|
||||
self.targets.append(target)
|
||||
|
||||
def start(self):
|
||||
pass
|
||||
|
||||
|
||||
class ObservedEvent:
|
||||
"""A real Event that also exposes when a waiter starts waiting."""
|
||||
|
||||
def __init__(self):
|
||||
self._event = threading.Event()
|
||||
self.wait_started = threading.Event()
|
||||
self.set_calls = 0
|
||||
|
||||
def set(self):
|
||||
self.set_calls += 1
|
||||
self._event.set()
|
||||
|
||||
def wait(self, timeout=None):
|
||||
self.wait_started.set()
|
||||
return self._event.wait(timeout)
|
||||
|
||||
def is_set(self):
|
||||
return self._event.is_set()
|
||||
|
||||
|
||||
class FakeReviewAgent:
|
||||
def __init__(self, **kwargs):
|
||||
self._session_messages = []
|
||||
|
||||
def run_conversation(self, **kwargs):
|
||||
pass
|
||||
|
||||
def interrupt(self, message=None):
|
||||
pass
|
||||
|
||||
def shutdown_memory_provider(self):
|
||||
pass
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
|
||||
def _bare_agent() -> AIAgent:
|
||||
agent = object.__new__(AIAgent)
|
||||
agent.model = "fake-model"
|
||||
@@ -31,6 +87,7 @@ def _bare_agent() -> AIAgent:
|
||||
agent._safe_print = lambda *_args, **_kwargs: None
|
||||
import threading as _threading
|
||||
agent._background_review_agent = None
|
||||
agent._background_review_run = None
|
||||
agent._background_review_lock = _threading.Lock()
|
||||
agent._active_children = []
|
||||
agent._active_children_lock = _threading.Lock()
|
||||
@@ -45,6 +102,89 @@ class ImmediateThread:
|
||||
self._target()
|
||||
|
||||
|
||||
def _install_live_turn_boundary(monkeypatch, on_boundary=None):
|
||||
import agent.conversation_loop as conversation_loop_module
|
||||
|
||||
def stop_at_boundary(*args, **kwargs):
|
||||
if on_boundary is not None:
|
||||
on_boundary()
|
||||
raise _TurnBoundaryReached
|
||||
|
||||
monkeypatch.setattr(
|
||||
conversation_loop_module,
|
||||
"build_turn_context",
|
||||
stop_at_boundary,
|
||||
)
|
||||
|
||||
|
||||
def _run_wrapped_live_turn_to_boundary(agent, result):
|
||||
try:
|
||||
result["return"] = AIAgent.run_conversation(
|
||||
agent,
|
||||
"next turn",
|
||||
task_id="live-task",
|
||||
)
|
||||
except _TurnBoundaryReached:
|
||||
result["boundary_reached"] = True
|
||||
except BaseException as exc: # surfaced in the test thread for a useful failure
|
||||
result["error"] = exc
|
||||
|
||||
|
||||
def _install_relay_recorder(monkeypatch, review_run=None):
|
||||
from agent import relay_runtime
|
||||
from hermes_cli.observability import relay_shared_metrics
|
||||
|
||||
calls = []
|
||||
|
||||
def review_acknowledged():
|
||||
return bool(review_run and review_run.request_done.is_set())
|
||||
|
||||
class RelayTurn:
|
||||
relay_enabled = True
|
||||
|
||||
class RecordingCoordinator:
|
||||
def acquire_conversation(self, **kwargs):
|
||||
calls.append(("acquire", review_acknowledged()))
|
||||
return object()
|
||||
|
||||
def begin_turn(self, lease, **kwargs):
|
||||
calls.append(("begin", review_acknowledged()))
|
||||
return RelayTurn()
|
||||
|
||||
def finish_logical_calls(self, turn, **kwargs):
|
||||
pass
|
||||
|
||||
def end_turn(self, turn, **kwargs):
|
||||
pass
|
||||
|
||||
def release_conversation(self, lease):
|
||||
pass
|
||||
|
||||
monkeypatch.setattr(
|
||||
relay_runtime,
|
||||
"SESSION_COORDINATOR",
|
||||
RecordingCoordinator(),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
relay_runtime,
|
||||
"current_profile_key",
|
||||
lambda: "/test-profile",
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
relay_shared_metrics,
|
||||
"start_task_run",
|
||||
lambda **kwargs: calls.append(
|
||||
("start_task_run", review_acknowledged())
|
||||
),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
relay_shared_metrics,
|
||||
"finish_task_run",
|
||||
lambda **kwargs: None,
|
||||
)
|
||||
return calls
|
||||
|
||||
|
||||
def test_background_review_shuts_down_memory_provider_before_close(monkeypatch):
|
||||
events = []
|
||||
|
||||
@@ -277,34 +417,19 @@ def test_background_review_explicit_focus_runs_even_in_subagent(monkeypatch):
|
||||
assert len(forks) == 1, "explicit focus review must run even in a subagent"
|
||||
|
||||
|
||||
def test_background_review_registers_on_active_children_for_interrupt(monkeypatch):
|
||||
"""The review fork must be added to the parent's ``_active_children`` so
|
||||
``AIAgent.interrupt()`` (which fans out to that list) can reach it, and
|
||||
to ``_background_review_agent`` so the NEXT live turn can proactively
|
||||
cancel a still-running review. Regression for the doubled-token-
|
||||
accounting / Ctrl+C-proof lockup that a review racing a new live turn
|
||||
against the same session_id/credentials can cause.
|
||||
"""
|
||||
def test_background_review_registers_before_start_runs_and_cleans_up(monkeypatch):
|
||||
"""The parent must own a unique review run before the worker can start."""
|
||||
seen = {}
|
||||
|
||||
class FakeReviewAgent:
|
||||
def __init__(self, **kwargs):
|
||||
self._session_messages = []
|
||||
|
||||
class RecordingReviewAgent(FakeReviewAgent):
|
||||
def run_conversation(self, **kwargs):
|
||||
# While run_conversation is "in flight", both tracking slots on
|
||||
# the parent must already point at this fork.
|
||||
seen["run"] = agent._background_review_run
|
||||
seen["active_children_during_run"] = list(agent._active_children)
|
||||
seen["background_review_agent_during_run"] = agent._background_review_agent
|
||||
|
||||
def shutdown_memory_provider(self):
|
||||
pass
|
||||
|
||||
def close(self):
|
||||
pass
|
||||
|
||||
monkeypatch.setattr(run_agent_module, "AIAgent", FakeReviewAgent)
|
||||
monkeypatch.setattr(run_agent_module.threading, "Thread", ImmediateThread)
|
||||
monkeypatch.setattr(run_agent_module, "AIAgent", RecordingReviewAgent)
|
||||
CapturingThread.targets = []
|
||||
monkeypatch.setattr(run_agent_module.threading, "Thread", CapturingThread)
|
||||
|
||||
agent = _bare_agent()
|
||||
|
||||
@@ -314,53 +439,333 @@ def test_background_review_registers_on_active_children_for_interrupt(monkeypatc
|
||||
review_memory=True,
|
||||
)
|
||||
|
||||
run = agent._background_review_run
|
||||
assert run is not None
|
||||
assert len(CapturingThread.targets) == 1
|
||||
assert not run.request_done.is_set()
|
||||
|
||||
observed_done = ObservedEvent()
|
||||
run.request_done = observed_done
|
||||
CapturingThread.targets[0]()
|
||||
|
||||
fork = seen["background_review_agent_during_run"]
|
||||
assert fork is not None
|
||||
assert seen["run"] is run
|
||||
assert seen["active_children_during_run"] == [fork]
|
||||
|
||||
# After the review completes, both tracking slots must be cleared —
|
||||
# otherwise a later interrupt() would try to cancel an already-closed
|
||||
# agent, or the next turn would wait on a review that no longer exists.
|
||||
assert observed_done.is_set()
|
||||
assert observed_done.set_calls == 1
|
||||
assert agent._background_review_run is None
|
||||
assert agent._background_review_agent is None
|
||||
assert agent._active_children == []
|
||||
|
||||
|
||||
def test_new_live_turn_cancels_still_running_background_review(monkeypatch):
|
||||
"""conversation_loop.run_conversation() must proactively interrupt a
|
||||
background review still in flight from a prior turn, rather than let the
|
||||
two race concurrently against the same session_id/credentials. This is
|
||||
the other half of the fix: registration alone only enables interrupt()
|
||||
propagation, it doesn't by itself stop the race — something has to
|
||||
actually call interrupt() at the start of the next turn.
|
||||
"""
|
||||
import agent.conversation_loop as conversation_loop_module
|
||||
def test_live_turn_waits_for_review_exit_before_relay_and_turn_context(monkeypatch):
|
||||
"""The outer production wrapper waits before same-session instrumentation."""
|
||||
review_entered = threading.Event()
|
||||
review_returned = threading.Event()
|
||||
allow_review_return = threading.Event()
|
||||
interrupted = threading.Event()
|
||||
boundary_reached = threading.Event()
|
||||
seen = {}
|
||||
|
||||
calls = []
|
||||
|
||||
class FakeReviewAgent:
|
||||
class BlockingReviewAgent(FakeReviewAgent):
|
||||
def interrupt(self, message=None):
|
||||
calls.append(message)
|
||||
seen["interrupt_message"] = message
|
||||
interrupted.set()
|
||||
|
||||
def run_conversation(self, **kwargs):
|
||||
review_entered.set()
|
||||
assert allow_review_return.wait(2.0)
|
||||
review_returned.set()
|
||||
|
||||
monkeypatch.setattr(run_agent_module, "AIAgent", BlockingReviewAgent)
|
||||
CapturingThread.targets = []
|
||||
monkeypatch.setattr(run_agent_module.threading, "Thread", CapturingThread)
|
||||
|
||||
agent = _bare_agent()
|
||||
agent._background_review_agent = FakeReviewAgent()
|
||||
AIAgent._spawn_background_review(
|
||||
agent,
|
||||
messages_snapshot=[{"role": "user", "content": "hello"}],
|
||||
review_memory=True,
|
||||
)
|
||||
run = agent._background_review_run
|
||||
assert run is not None
|
||||
observed_done = ObservedEvent()
|
||||
run.request_done = observed_done
|
||||
|
||||
# Invoke just the cancellation snippet in isolation via the same
|
||||
# attribute contract run_conversation() reads, to avoid dragging in the
|
||||
# rest of the turn machinery (network calls, tool setup, etc.) that
|
||||
# isn't relevant to this regression.
|
||||
_pending_review = getattr(agent, "_background_review_agent", None)
|
||||
assert _pending_review is not None
|
||||
_pending_review.interrupt("superseded by a new live turn")
|
||||
monkeypatch.setattr(run_agent_module.threading, "Thread", _REAL_THREAD)
|
||||
worker = _REAL_THREAD(target=CapturingThread.targets[0], daemon=True)
|
||||
worker.start()
|
||||
assert review_entered.wait(2.0)
|
||||
|
||||
assert calls == ["superseded by a new live turn"]
|
||||
def on_boundary():
|
||||
seen["review_returned_at_boundary"] = review_returned.is_set()
|
||||
boundary_reached.set()
|
||||
|
||||
_install_live_turn_boundary(monkeypatch, on_boundary)
|
||||
relay_calls = _install_relay_recorder(monkeypatch, run)
|
||||
live_result = {}
|
||||
live = _REAL_THREAD(
|
||||
target=_run_wrapped_live_turn_to_boundary,
|
||||
args=(agent, live_result),
|
||||
daemon=True,
|
||||
)
|
||||
live.start()
|
||||
|
||||
assert interrupted.wait(2.0)
|
||||
wait_started = observed_done.wait_started.wait(2.0)
|
||||
relay_calls_before_ack = list(relay_calls)
|
||||
allow_review_return.set()
|
||||
worker.join(timeout=2.0)
|
||||
live.join(timeout=2.0)
|
||||
|
||||
assert not worker.is_alive()
|
||||
assert not live.is_alive()
|
||||
assert wait_started
|
||||
assert relay_calls_before_ack == []
|
||||
assert relay_calls == [
|
||||
("acquire", True),
|
||||
("begin", True),
|
||||
("start_task_run", True),
|
||||
]
|
||||
assert boundary_reached.is_set()
|
||||
assert seen["interrupt_message"] == "superseded by a new live turn"
|
||||
assert seen["review_returned_at_boundary"] is True
|
||||
assert live_result == {"boundary_reached": True}
|
||||
|
||||
|
||||
def test_live_turn_cancels_review_during_startup_before_provider(monkeypatch):
|
||||
"""A review cancelled before its worker runs must never call its provider."""
|
||||
provider_calls = []
|
||||
boundary_reached = threading.Event()
|
||||
|
||||
class RecordingReviewAgent(FakeReviewAgent):
|
||||
def run_conversation(self, **kwargs):
|
||||
provider_calls.append(kwargs)
|
||||
|
||||
monkeypatch.setattr(run_agent_module, "AIAgent", RecordingReviewAgent)
|
||||
CapturingThread.targets = []
|
||||
monkeypatch.setattr(run_agent_module.threading, "Thread", CapturingThread)
|
||||
|
||||
agent = _bare_agent()
|
||||
AIAgent._spawn_background_review(
|
||||
agent,
|
||||
messages_snapshot=[{"role": "user", "content": "hello"}],
|
||||
review_memory=True,
|
||||
)
|
||||
run = agent._background_review_run
|
||||
assert run is not None
|
||||
|
||||
_install_live_turn_boundary(monkeypatch, boundary_reached.set)
|
||||
relay_calls = _install_relay_recorder(monkeypatch, run)
|
||||
live_result = {}
|
||||
live = _REAL_THREAD(
|
||||
target=_run_wrapped_live_turn_to_boundary,
|
||||
args=(agent, live_result),
|
||||
daemon=True,
|
||||
)
|
||||
live.start()
|
||||
assert run.cancel_requested.wait(2.0)
|
||||
|
||||
worker = _REAL_THREAD(target=CapturingThread.targets[0], daemon=True)
|
||||
worker.start()
|
||||
worker.join(timeout=2.0)
|
||||
live.join(timeout=2.0)
|
||||
|
||||
assert not worker.is_alive()
|
||||
assert not live.is_alive()
|
||||
assert boundary_reached.is_set()
|
||||
assert provider_calls == []
|
||||
assert run.request_done.is_set()
|
||||
assert relay_calls == [
|
||||
("acquire", True),
|
||||
("begin", True),
|
||||
("start_task_run", True),
|
||||
]
|
||||
assert live_result == {"boundary_reached": True}
|
||||
|
||||
|
||||
def test_live_turn_proceeds_when_review_acknowledgement_times_out(monkeypatch):
|
||||
"""A broken review abort path must not block the foreground indefinitely.
|
||||
The live turn proceeds after the bounded wait, retaining foreground priority.
|
||||
"""
|
||||
import time
|
||||
|
||||
import agent.background_review as background_review_module
|
||||
|
||||
review_entered = threading.Event()
|
||||
interrupt_entered = threading.Event()
|
||||
interrupt_returned = threading.Event()
|
||||
allow_interrupt_return = threading.Event()
|
||||
allow_review_return = threading.Event()
|
||||
|
||||
class WedgedReviewAgent(FakeReviewAgent):
|
||||
def run_conversation(self, **kwargs):
|
||||
review_entered.set()
|
||||
allow_review_return.wait(5.0)
|
||||
|
||||
def interrupt(self, message=None):
|
||||
interrupt_entered.set()
|
||||
allow_interrupt_return.wait(5.0)
|
||||
interrupt_returned.set()
|
||||
|
||||
monkeypatch.setattr(run_agent_module, "AIAgent", WedgedReviewAgent)
|
||||
CapturingThread.targets = []
|
||||
monkeypatch.setattr(run_agent_module.threading, "Thread", CapturingThread)
|
||||
monkeypatch.setattr(
|
||||
background_review_module,
|
||||
"_BACKGROUND_REVIEW_CANCEL_TIMEOUT_SECONDS",
|
||||
0.01,
|
||||
raising=False,
|
||||
)
|
||||
|
||||
agent = _bare_agent()
|
||||
AIAgent._spawn_background_review(
|
||||
agent,
|
||||
messages_snapshot=[{"role": "user", "content": "hello"}],
|
||||
review_memory=True,
|
||||
)
|
||||
run = agent._background_review_run
|
||||
assert run is not None
|
||||
monkeypatch.setattr(run_agent_module.threading, "Thread", _REAL_THREAD)
|
||||
worker = _REAL_THREAD(target=CapturingThread.targets[0], daemon=True)
|
||||
worker.start()
|
||||
assert review_entered.wait(2.0)
|
||||
|
||||
boundary_calls = []
|
||||
_install_live_turn_boundary(
|
||||
monkeypatch, lambda: boundary_calls.append(True)
|
||||
)
|
||||
relay_calls = _install_relay_recorder(monkeypatch, run)
|
||||
|
||||
started = time.monotonic()
|
||||
live_result = {}
|
||||
live = _REAL_THREAD(
|
||||
target=_run_wrapped_live_turn_to_boundary,
|
||||
args=(agent, live_result),
|
||||
daemon=True,
|
||||
)
|
||||
live.start()
|
||||
live.join(timeout=5.0)
|
||||
|
||||
elapsed = time.monotonic() - started
|
||||
|
||||
allow_interrupt_return.set()
|
||||
allow_review_return.set()
|
||||
worker.join(timeout=2.0)
|
||||
|
||||
assert elapsed < 2.0
|
||||
assert interrupt_entered.is_set()
|
||||
assert interrupt_returned.wait(2.0)
|
||||
assert not worker.is_alive()
|
||||
assert not live.is_alive()
|
||||
# Foreground retains priority: Relay/turn-context proceed even though
|
||||
# the review did not acknowledge within the bounded deadline.
|
||||
assert boundary_calls == [True]
|
||||
assert live_result == {"boundary_reached": True}
|
||||
assert relay_calls == [
|
||||
("acquire", False),
|
||||
("begin", False),
|
||||
("start_task_run", False),
|
||||
]
|
||||
assert agent.session_id == "test-session"
|
||||
|
||||
|
||||
def test_live_turn_interrupts_legacy_review_but_keeps_foreground_priority(monkeypatch):
|
||||
"""Legacy stubs are interrupted without turning review into a user blocker."""
|
||||
interrupts = []
|
||||
interrupt_called = threading.Event()
|
||||
|
||||
class LegacyReviewAgent:
|
||||
def interrupt(self, message=None):
|
||||
interrupts.append(message)
|
||||
interrupt_called.set()
|
||||
|
||||
agent = _bare_agent()
|
||||
del agent._background_review_run
|
||||
agent._background_review_agent = LegacyReviewAgent()
|
||||
boundary_calls = []
|
||||
_install_live_turn_boundary(
|
||||
monkeypatch, lambda: boundary_calls.append(True)
|
||||
)
|
||||
relay_calls = _install_relay_recorder(monkeypatch)
|
||||
|
||||
live_result = {}
|
||||
live = _REAL_THREAD(
|
||||
target=_run_wrapped_live_turn_to_boundary,
|
||||
args=(agent, live_result),
|
||||
daemon=True,
|
||||
)
|
||||
live.start()
|
||||
live.join(timeout=5.0)
|
||||
|
||||
assert interrupt_called.wait(2.0)
|
||||
assert interrupts == ["superseded by a new live turn"]
|
||||
assert not live.is_alive()
|
||||
assert boundary_calls == [True]
|
||||
assert live_result == {"boundary_reached": True}
|
||||
assert relay_calls == [
|
||||
("acquire", False),
|
||||
("begin", False),
|
||||
("start_task_run", False),
|
||||
]
|
||||
assert agent.session_id == "test-session"
|
||||
|
||||
|
||||
def test_stale_review_cleanup_cannot_clear_or_signal_newer_review(monkeypatch):
|
||||
"""A retired worker's late cleanup must be scoped to its own run identity."""
|
||||
first_cleanup_entered = threading.Event()
|
||||
allow_first_cleanup = threading.Event()
|
||||
instance_count = 0
|
||||
|
||||
class BlockingCleanupReviewAgent(FakeReviewAgent):
|
||||
def __init__(self, **kwargs):
|
||||
nonlocal instance_count
|
||||
super().__init__(**kwargs)
|
||||
self.index = instance_count
|
||||
instance_count += 1
|
||||
|
||||
def shutdown_memory_provider(self):
|
||||
if self.index == 0:
|
||||
first_cleanup_entered.set()
|
||||
assert allow_first_cleanup.wait(2.0)
|
||||
|
||||
monkeypatch.setattr(run_agent_module, "AIAgent", BlockingCleanupReviewAgent)
|
||||
CapturingThread.targets = []
|
||||
monkeypatch.setattr(run_agent_module.threading, "Thread", CapturingThread)
|
||||
|
||||
agent = _bare_agent()
|
||||
AIAgent._spawn_background_review(
|
||||
agent,
|
||||
messages_snapshot=[{"role": "user", "content": "first"}],
|
||||
review_memory=True,
|
||||
)
|
||||
first_run = agent._background_review_run
|
||||
first_worker = _REAL_THREAD(target=CapturingThread.targets[0], daemon=True)
|
||||
first_worker.start()
|
||||
assert first_cleanup_entered.wait(2.0)
|
||||
assert first_run.request_done.is_set()
|
||||
|
||||
AIAgent._spawn_background_review(
|
||||
agent,
|
||||
messages_snapshot=[{"role": "user", "content": "second"}],
|
||||
review_memory=True,
|
||||
)
|
||||
second_run = agent._background_review_run
|
||||
second_target = CapturingThread.targets[1]
|
||||
assert second_run is not first_run
|
||||
assert not second_run.request_done.is_set()
|
||||
|
||||
allow_first_cleanup.set()
|
||||
first_worker.join(timeout=2.0)
|
||||
|
||||
assert not first_worker.is_alive()
|
||||
assert agent._background_review_run is second_run
|
||||
assert not second_run.request_done.is_set()
|
||||
|
||||
second_target()
|
||||
assert second_run.request_done.is_set()
|
||||
assert agent._background_review_run is None
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# memory_notifications mode: off | on | verbose
|
||||
|
||||
@@ -0,0 +1,394 @@
|
||||
"""Tests for native compaction summary retention during pre-checkpoint pruning (#90975).
|
||||
|
||||
``prune_pre_checkpoint_items`` previously dropped every pre-checkpoint item
|
||||
whose ``role`` was not ``"user"`` — which silently deleted Hermes' own local
|
||||
compression summaries (``role="assistant"``) from the wire on every native
|
||||
compaction turn. These tests cover the fix's summary retention path, its
|
||||
reliance on the canonical ``agent.context_compressor`` provenance check (not
|
||||
an ad-hoc heuristic), whole-or-drop truncation, and idempotency.
|
||||
"""
|
||||
|
||||
from agent.context_compressor import (
|
||||
COMPRESSED_SUMMARY_METADATA_KEY,
|
||||
ContextCompressor,
|
||||
SUMMARY_PREFIX,
|
||||
_MERGED_PRIOR_CONTEXT_HEADER,
|
||||
_MERGED_SUMMARY_DELIMITER,
|
||||
_SUMMARY_END_MARKER,
|
||||
)
|
||||
from agent.native_compaction import (
|
||||
_extract_item_text,
|
||||
_is_summary_item,
|
||||
prune_pre_checkpoint_items,
|
||||
)
|
||||
|
||||
|
||||
def _standalone_summary_content(body: str = "## Active Task\nstuff") -> str:
|
||||
return f"{SUMMARY_PREFIX}\n{body}\n\n{_SUMMARY_END_MARKER}"
|
||||
|
||||
|
||||
def _merged_summary_content(tail: str = "preserved prior turn") -> str:
|
||||
return (
|
||||
f"{_MERGED_PRIOR_CONTEXT_HEADER}\n{tail}\n\n"
|
||||
f"{_MERGED_SUMMARY_DELIMITER}\n\n"
|
||||
f"{SUMMARY_PREFIX}\nbody\n\n{_SUMMARY_END_MARKER}"
|
||||
)
|
||||
|
||||
|
||||
class TestIsSummaryItemCanonical:
|
||||
"""`_is_summary_item` must delegate to the canonical provenance check —
|
||||
exact metadata flag or the canonical prefix classifier — never an
|
||||
ad-hoc heuristic (#90975 blocking review)."""
|
||||
|
||||
def test_truthy_metadata_flag_detected(self):
|
||||
assert _is_summary_item({COMPRESSED_SUMMARY_METADATA_KEY: True}) is True
|
||||
|
||||
def test_standalone_content_detected_without_metadata(self):
|
||||
# The wire sanitizers strip underscore keys, so content-only
|
||||
# detection must still work on the canonical prefix.
|
||||
assert _is_summary_item({"role": "assistant", "content": _standalone_summary_content()}) is True
|
||||
|
||||
def test_merged_content_detected_without_metadata(self):
|
||||
assert _is_summary_item({"role": "assistant", "content": _merged_summary_content()}) is True
|
||||
|
||||
def test_malformed_inputs_are_not_summaries(self):
|
||||
assert _is_summary_item(None) is False
|
||||
assert _is_summary_item(123) is False
|
||||
assert _is_summary_item({}) is False
|
||||
|
||||
|
||||
class TestIsSummaryItemNegativeWitnesses:
|
||||
"""Content that merely resembles a summary must never be promoted to
|
||||
durable retained history — that is authority drift (#90975 blocking
|
||||
review, required item 4)."""
|
||||
|
||||
def test_summary_heading_in_ordinary_user_text_is_not_a_summary(self):
|
||||
item = {"role": "user", "content": "## Summary\nplease summarize the PR for me"}
|
||||
assert _is_summary_item(item) is False
|
||||
|
||||
def test_false_valued_metadata_flag_is_not_a_summary(self):
|
||||
item = {"role": "assistant", "content": "hi", COMPRESSED_SUMMARY_METADATA_KEY: False}
|
||||
assert _is_summary_item(item) is False
|
||||
|
||||
def test_arbitrary_underscore_summary_key_is_not_a_summary(self):
|
||||
item = {"role": "assistant", "content": "hi", "_my_custom_summary_flag": True}
|
||||
assert _is_summary_item(item) is False
|
||||
|
||||
def test_non_hermes_assistant_content_is_not_a_summary(self):
|
||||
item = {"role": "assistant", "content": "Conversation Summary: I finished the task."}
|
||||
assert _is_summary_item(item) is False
|
||||
|
||||
|
||||
class TestExtractItemTextVariations:
|
||||
def test_string_content(self):
|
||||
assert _extract_item_text({"content": "Hello world"}) == "Hello world"
|
||||
|
||||
def test_multipart_list_content(self):
|
||||
item = {
|
||||
"content": [
|
||||
{"type": "input_text", "text": "Part 1"},
|
||||
{"type": "text", "text": "Part 2"},
|
||||
{"type": "other", "output_text": "Part 3"},
|
||||
]
|
||||
}
|
||||
assert _extract_item_text(item) == "Part 1 Part 2 Part 3"
|
||||
|
||||
def test_output_text_fallback(self):
|
||||
assert _extract_item_text({"output_text": "Output fallback"}) == "Output fallback"
|
||||
|
||||
def test_malformed_or_empty(self):
|
||||
assert _extract_item_text({"content": None}) is None
|
||||
assert _extract_item_text({"content": []}) is None
|
||||
assert _extract_item_text(None) is None
|
||||
assert _extract_item_text("string_item") is None
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsRetainsSummaries:
|
||||
def test_retains_summary_and_user_in_original_order(self):
|
||||
summary_content = _standalone_summary_content("Step 1 complete")
|
||||
items = [
|
||||
{"role": "user", "content": "User Ask 1"},
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"role": "user", "content": "User Ask 2"},
|
||||
{"role": "assistant", "content": "Normal chatter to prune"},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "User Ask 3"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(items, retained_user_token_budget=1000)
|
||||
|
||||
assert pruned[0]["type"] == "compaction"
|
||||
contents = [m.get("content") for m in pruned[1:]]
|
||||
assert contents == [
|
||||
"User Ask 1",
|
||||
summary_content,
|
||||
"User Ask 2",
|
||||
"User Ask 3",
|
||||
]
|
||||
|
||||
def test_role_agnostic_retention_does_not_touch_user_budget(self):
|
||||
summary_content = _standalone_summary_content("x" * 2000)
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"role": "user", "content": "short ask"},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(
|
||||
items, retained_user_token_budget=10, retained_summary_token_budget=10_000
|
||||
)
|
||||
|
||||
contents = [m.get("content") for m in pruned]
|
||||
assert summary_content in contents
|
||||
assert "short ask" in contents
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsSummaryBudget:
|
||||
def test_oversized_summary_is_dropped_whole_not_sliced(self):
|
||||
"""A summary that cannot fit the remaining budget is dropped
|
||||
entirely rather than character-sliced (#90975 blocking review,
|
||||
required item 3): slicing can corrupt the handoff prefix / end
|
||||
marker that keeps the summary non-active."""
|
||||
long_summary = _standalone_summary_content("Summary line " * 500)
|
||||
items = [
|
||||
{"role": "assistant", "content": long_summary, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(items, retained_summary_token_budget=100)
|
||||
|
||||
assert not any(m.get(COMPRESSED_SUMMARY_METADATA_KEY) for m in pruned)
|
||||
|
||||
def test_summary_that_fits_budget_is_retained_whole(self):
|
||||
summary_content = _standalone_summary_content("short body")
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(items, retained_summary_token_budget=10_000)
|
||||
|
||||
retained = [m for m in pruned if m.get(COMPRESSED_SUMMARY_METADATA_KEY)]
|
||||
assert len(retained) == 1
|
||||
assert retained[0]["content"] == summary_content
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsIdempotency:
|
||||
def test_duplicate_summary_text_is_not_retained_twice(self):
|
||||
"""A repeated checkpoint sequence can leave the same summary text
|
||||
present at more than one pre-checkpoint position; retention must
|
||||
stay idempotent rather than duplicate it (#90975 blocking review,
|
||||
required item 5)."""
|
||||
summary_content = _standalone_summary_content("same body")
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"role": "user", "content": "mid ask"},
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
|
||||
pruned = prune_pre_checkpoint_items(items)
|
||||
|
||||
matches = [m for m in pruned if m.get("content") == summary_content]
|
||||
assert len(matches) == 1
|
||||
|
||||
def test_re_pruning_an_already_pruned_result_is_stable(self):
|
||||
summary_content = _standalone_summary_content("stable body")
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"role": "user", "content": "ask"},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
]
|
||||
|
||||
once = prune_pre_checkpoint_items(items)
|
||||
twice = prune_pre_checkpoint_items(once)
|
||||
assert once == twice
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsLiveCompressorEmissions:
|
||||
"""Exercise the real ``ContextCompressor`` marker renderer instead of a
|
||||
hand-built stand-in, for both standalone and merge-into-tail shapes
|
||||
(#90975 blocking review, required item 5)."""
|
||||
|
||||
def test_standalone_live_marker_is_retained(self):
|
||||
rendered = ContextCompressor._render_micro_marker_content("Live handoff body")
|
||||
assert ContextCompressor.classify_summary_content(rendered) == "standalone"
|
||||
|
||||
items = [
|
||||
{"role": "assistant", "content": rendered, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
pruned = prune_pre_checkpoint_items(items)
|
||||
assert any(m.get("content") == rendered for m in pruned)
|
||||
|
||||
def test_merged_tail_summary_is_retained_and_classified_merged(self):
|
||||
merged = _merged_summary_content("earlier preserved turn text")
|
||||
assert ContextCompressor.classify_summary_content(merged) == "merged"
|
||||
|
||||
items = [
|
||||
{"role": "assistant", "content": merged, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob_cp"},
|
||||
{"role": "user", "content": "Ask"},
|
||||
]
|
||||
pruned = prune_pre_checkpoint_items(items)
|
||||
assert any(m.get("content") == merged for m in pruned)
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsEnableSummaryRetentionToggle:
|
||||
def test_disabling_summary_retention_drops_pre_checkpoint_summaries(self):
|
||||
summary_content = _standalone_summary_content("Old")
|
||||
items = [
|
||||
{"role": "assistant", "content": summary_content, COMPRESSED_SUMMARY_METADATA_KEY: True},
|
||||
{"type": "compaction", "encrypted_content": "blob"},
|
||||
{"role": "user", "content": "New ask"},
|
||||
]
|
||||
|
||||
pruned_disabled = prune_pre_checkpoint_items(items, enable_summary_retention=False)
|
||||
contents = [m.get("content") for m in pruned_disabled]
|
||||
assert summary_content not in contents
|
||||
|
||||
|
||||
def _checkpoint_message(item_id: str = "rs_cp1", blob: str = "cp_blob_1"):
|
||||
"""An assistant message carrying a replayable native-compaction checkpoint."""
|
||||
return {
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"codex_reasoning_items": [
|
||||
{"type": "compaction", "encrypted_content": blob, "id": item_id},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
class TestChatMessagesToResponsesInputSummaryCarrierLoss:
|
||||
"""Adapter-level witnesses for the second blocking review (#90976):
|
||||
``prune_pre_checkpoint_items`` only ever saw whatever ``_is_summary_item``
|
||||
could recover from an already-converted Responses ``item`` — but two
|
||||
real merge-into-tail carrier shapes lose or shadow the summary content
|
||||
during ``_chat_messages_to_responses_input`` itself, *before* pruning
|
||||
ever runs:
|
||||
|
||||
* a tool-result carrier becomes a typed ``function_call_output`` (no
|
||||
``content``/``role`` survive the conversion at all), and
|
||||
* an assistant carrier with a stale ``codex_message_items`` sidecar
|
||||
replays the pre-merge exact message item instead of the rewritten
|
||||
(summary-bearing) ``content``.
|
||||
|
||||
These feed real chat messages, shaped exactly the way
|
||||
``ContextCompressor.compress()`` merge-into-tail produces them (same
|
||||
``COMPRESSED_SUMMARY_METADATA_KEY`` stamp, same merge delimiters/end
|
||||
marker), through the real ``_chat_messages_to_responses_input`` with a
|
||||
replayed checkpoint — not a hand-built Responses item passed straight
|
||||
to the pruner.
|
||||
"""
|
||||
|
||||
def test_tool_result_merge_carrier_summary_survives_the_adapter(self):
|
||||
from agent.codex_responses_adapter import _chat_messages_to_responses_input
|
||||
|
||||
merged = _merged_summary_content("preserved tool context")
|
||||
messages = [
|
||||
{"role": "user", "content": "please do the thing"},
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [{
|
||||
"id": "call_1",
|
||||
"type": "function",
|
||||
"function": {"name": "do_thing", "arguments": "{}"},
|
||||
}],
|
||||
},
|
||||
{
|
||||
"role": "tool",
|
||||
"tool_call_id": "call_1",
|
||||
"content": merged,
|
||||
COMPRESSED_SUMMARY_METADATA_KEY: True,
|
||||
},
|
||||
_checkpoint_message(),
|
||||
{"role": "user", "content": "next ask after checkpoint"},
|
||||
]
|
||||
|
||||
items = _chat_messages_to_responses_input(
|
||||
messages, native_compaction_eligible=True,
|
||||
)
|
||||
|
||||
# The summary survives, exactly once, as a plain message item —
|
||||
# never as a `function_call_output` (which the pruner cannot see,
|
||||
# and which would orphan the dropped `function_call` it used to
|
||||
# pair with).
|
||||
assert not any(
|
||||
isinstance(it, dict) and it.get("type") == "function_call_output"
|
||||
for it in items
|
||||
)
|
||||
matches = [
|
||||
it for it in items
|
||||
if isinstance(it, dict) and _extract_item_text(it) == merged
|
||||
]
|
||||
assert len(matches) == 1
|
||||
assert matches[0].get("type") != "function_call_output"
|
||||
|
||||
# And the newest checkpoint still leads the wire.
|
||||
assert items[0].get("type") == "compaction"
|
||||
|
||||
def test_assistant_merge_carrier_with_stale_replay_summary_survives(self):
|
||||
from agent.codex_responses_adapter import _chat_messages_to_responses_input
|
||||
|
||||
merged = _merged_summary_content("preserved assistant context")
|
||||
messages = [
|
||||
{"role": "user", "content": "question"},
|
||||
{
|
||||
"role": "assistant",
|
||||
# Rewritten by the compressor merge — this is what must
|
||||
# reach the wire.
|
||||
"content": merged,
|
||||
COMPRESSED_SUMMARY_METADATA_KEY: True,
|
||||
# Stale sidecar captured BEFORE the merge rewrote the
|
||||
# content above. The exact-replay path prefers this over
|
||||
# `content` for prefix-cache continuity, which is exactly
|
||||
# what shadows the summary (#90976).
|
||||
"codex_message_items": [{
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"id": "msg_stale_1",
|
||||
"status": "completed",
|
||||
"content": [{"type": "output_text", "text": "stale pre-merge answer"}],
|
||||
}],
|
||||
},
|
||||
_checkpoint_message(),
|
||||
{"role": "user", "content": "next ask"},
|
||||
]
|
||||
|
||||
items = _chat_messages_to_responses_input(
|
||||
messages, native_compaction_eligible=True,
|
||||
)
|
||||
|
||||
assert not any(
|
||||
isinstance(it, dict) and _extract_item_text(it) == "stale pre-merge answer"
|
||||
for it in items
|
||||
)
|
||||
matches = [
|
||||
it for it in items
|
||||
if isinstance(it, dict) and _extract_item_text(it) == merged
|
||||
]
|
||||
assert len(matches) == 1
|
||||
assert items[0].get("type") == "compaction"
|
||||
|
||||
|
||||
class TestPrunePreCheckpointItemsMalformedInputs:
|
||||
def test_handles_none_non_dict_and_empty_items_safely(self):
|
||||
assert prune_pre_checkpoint_items(None) is None
|
||||
assert prune_pre_checkpoint_items([]) == []
|
||||
|
||||
items = [
|
||||
None,
|
||||
123,
|
||||
"raw_string",
|
||||
{"role": "user", "content": "Valid user ask"},
|
||||
{"type": "compaction", "encrypted_content": "blob"},
|
||||
]
|
||||
pruned = prune_pre_checkpoint_items(items)
|
||||
assert len(pruned) == 2
|
||||
assert pruned[0]["type"] == "compaction"
|
||||
assert pruned[1]["content"] == "Valid user ask"
|
||||
@@ -8,7 +8,7 @@ resolution-markers = [
|
||||
]
|
||||
|
||||
[options]
|
||||
exclude-newer = "2026-08-04T07:06:16.774262776Z"
|
||||
exclude-newer = "0001-01-01T00:00:00Z" # This has no effect and is included for backwards compatibility when using relative exclude-newer values.
|
||||
exclude-newer-span = "P14D"
|
||||
|
||||
[options.exclude-newer-package]
|
||||
@@ -1569,7 +1569,7 @@ wheels = [
|
||||
|
||||
[[package]]
|
||||
name = "hermes-agent"
|
||||
version = "0.20.4"
|
||||
version = "0.20.5"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "certifi" },
|
||||
|
||||
@@ -176,7 +176,9 @@ agent↔Nous wire contract lives in `docs/chronos-managed-cron-contract.md`.
|
||||
Each cron job runs in a completely fresh agent session:
|
||||
|
||||
- No conversation history from previous runs
|
||||
- No memory of previous cron executions (unless persisted to memory/files)
|
||||
- No memory of previous cron executions (persistent memory — MEMORY.md /
|
||||
USER.md — does load, like any other agent run, so durable preferences and
|
||||
facts carry over; per-run conversation context does not)
|
||||
- The prompt must be self-contained — cron jobs cannot ask clarifying questions
|
||||
- The `cronjob` toolset is disabled (recursion guard)
|
||||
|
||||
|
||||
@@ -80,6 +80,16 @@ You can pass `--keep-stash` to a terminal `hermes update` too if you want the sa
|
||||
|
||||
Want to know if an update is available before pulling? Run `hermes update --check` — it fetches and compares commits against `origin/main`. No files are modified, no gateway is restarted. Useful in scripts and cron jobs that gate on "is there an update".
|
||||
|
||||
### Fleet preview: `hermes update --plan`
|
||||
|
||||
Before updating a machine that runs several profiles or services, `hermes update --plan` prints the full update plan without changing anything: the install kind (git checkout, Docker image, Nix/apt managed), every running Hermes service across all profiles with its supervisor (systemd, launchd, manual) and the code version it is actually serving, and the restart mechanism each one will get. On image- or package-managed installs the plan reports that the install is not updatable in place and names the right update command instead. Read-only and safe on a live fleet.
|
||||
|
||||
The same inventory is embedded in every real update's receipt (`~/.hermes/logs/update_receipts/`), so after an update you can compare what the updater saw against what it did.
|
||||
|
||||
### Update receipts and the fleet version check
|
||||
|
||||
Every `hermes update` run writes a machine-readable receipt to `~/.hermes/logs/update_receipts/` (last 20 kept, `latest.json` always points at the most recent): the pre-update fleet plan, each step taken, anything skipped and why, the gateway restart outcome, and the final fleet version matrix. After the restart phase the updater compares each live gateway's running code against the freshly updated checkout and prints a per-profile matrix — a gateway still serving pre-update code is reported loudly with the exact restart command, and the update exits non-zero so automation never treats a mixed-version fleet as healthy.
|
||||
|
||||
### Full pre-update backup: `--backup`
|
||||
|
||||
For high-value profiles (production gateways, shared team installs) you can opt into a full pre-pull backup of `HERMES_HOME` (config, auth, sessions, skills, pairing):
|
||||
|
||||
@@ -123,7 +123,7 @@ Otherwise, provide a concise summary of the activity." --name "Repo watcher" --d
|
||||
```
|
||||
|
||||
:::warning Self-Contained Prompts
|
||||
Notice how the prompt includes the exact `gh` commands. The cron agent has no memory of previous runs or your preferences — spell everything out.
|
||||
Notice how the prompt includes the exact `gh` commands. The cron agent has no conversation history from previous runs — spell everything out. (Persistent memory does load, so durable preferences saved to MEMORY.md carry over, but don't rely on it for job-critical details.)
|
||||
:::
|
||||
|
||||
---
|
||||
|
||||
@@ -1730,7 +1730,7 @@ hermes completion fish > ~/.config/fish/completions/hermes.fish
|
||||
## `hermes update`
|
||||
|
||||
```bash
|
||||
hermes update [--gateway] [--check] [--no-backup] [--backup] [--yes]
|
||||
hermes update [--gateway] [--check] [--plan] [--no-backup] [--backup] [--yes]
|
||||
```
|
||||
|
||||
Pulls the latest `hermes-agent` code and reinstalls dependencies in the managed venv, then re-runs the post-install hooks (MCP servers, skills sync, completion install). Safe to run on a live install. Use `--check` to see whether your checkout is behind `origin/main` without installing.
|
||||
@@ -1741,6 +1741,7 @@ Pulls the latest `hermes-agent` code and reinstalls dependencies in the managed
|
||||
|--------|-------------|
|
||||
| `--gateway` | Internal mode used by the messaging `/update` command. Uses file-based IPC for prompts and progress streaming instead of reading from terminal stdin. Not a gateway restart flag. |
|
||||
| `--check` | Check whether an update is available without pulling, installing dependencies, or restarting anything. |
|
||||
| `--plan` | Print the update plan and exit without changing anything: install kind (git/Docker/Nix/apt), every running Hermes service across all profiles with its supervisor and running code version, and how each will be restarted. On image- or package-managed installs, reports the correct external update command instead. Read-only. |
|
||||
| `--no-backup` | Skip all pre-update backups for this run (both the quick state snapshot and the full zip), regardless of `updates.pre_update_backup`. |
|
||||
| `--backup` | Force a **full** pre-update backup for this run: the quick state snapshot plus a complete zip of `HERMES_HOME` (config, auth, sessions, skills, pairing data). The default mode is `quick` — a lightweight state snapshot only. Set the permanent mode via `updates.pre_update_backup: quick | full | off` in `config.yaml`. |
|
||||
| `--yes`, `-y` | Assume yes for interactive prompts such as config migration and stash restore. API-key entry is skipped; run `hermes config migrate` separately for those. |
|
||||
@@ -1748,6 +1749,7 @@ Pulls the latest `hermes-agent` code and reinstalls dependencies in the managed
|
||||
Additional behavior:
|
||||
|
||||
- **Gateway restart.** After a successful update, Hermes attempts to restart all running gateway profiles automatically so they pick up the new code. Use `hermes gateway restart` when you want to restart a gateway without applying an update.
|
||||
- **Update receipts + fleet version check.** Every run writes a machine-readable receipt to `~/.hermes/logs/update_receipts/` (pre-update fleet plan, steps, skips with reasons, restart outcome; `latest.json` points at the newest). After the restart phase the updater verifies each live gateway's running code against the updated checkout and prints a per-profile version matrix; a gateway still on pre-update code fails the update (exit 1) with the exact restart command.
|
||||
- **Local source changes.** For git installs, dirty tracked files and untracked files are auto-stashed before branch checkout or pull (`git stash push --include-untracked`). Interactive terminal updates ask before restoring the stash. Non-interactive updates restore it by default; set `updates.non_interactive_local_changes: discard` only on managed installs where local source edits should be thrown away after a successful pull. If stash restore conflicts or the pull fails, the stash is left in place for manual recovery.
|
||||
- **npm lockfile churn.** Before stashing or switching branches, Hermes makes a best-effort cleanup of tracked `package-lock.json` diffs produced by npm install/build steps. Commit or manually stash intentional lockfile edits before running `hermes update`.
|
||||
- **Pairing data snapshot.** Even when `--backup` is off, `hermes update` takes a lightweight snapshot of `~/.hermes/pairing/` and the Feishu comment rules before `git pull`. You can roll it back with `hermes backup restore --state pre-update` if a pull rewrites a file you were editing.
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"version": 1,
|
||||
"updated_at": "2026-08-21T03:50:21Z",
|
||||
"updated_at": "2026-08-21T11:29:05Z",
|
||||
"metadata": {
|
||||
"source": "hermes-agent repo",
|
||||
"docs": "https://hermes-agent.nousresearch.com/docs/reference/model-catalog"
|
||||
@@ -162,11 +162,15 @@
|
||||
"description": "free"
|
||||
},
|
||||
{
|
||||
"id": "poolside/laguna-m.1:free",
|
||||
"id": "z-ai/glm-5.2:free",
|
||||
"description": "free"
|
||||
},
|
||||
{
|
||||
"id": "tencent/hy3:free",
|
||||
"id": "poolside/laguna-s-2.1:free",
|
||||
"description": "free"
|
||||
},
|
||||
{
|
||||
"id": "poolside/laguna-xs-2.1:free",
|
||||
"description": "free"
|
||||
},
|
||||
{
|
||||
@@ -178,7 +182,7 @@
|
||||
"description": "free"
|
||||
},
|
||||
{
|
||||
"id": "inclusionai/ring-2.6-1t:free",
|
||||
"id": "nvidia/nemotron-3.5-lightning:free",
|
||||
"description": "free"
|
||||
}
|
||||
]
|
||||
|
||||
Reference in New Issue
Block a user