Merge upstream main (afc3d9d34c): refresh before review

One conflict: upstream 6e7c7c7da9 replaced bot-mode-closed-chat-stays-closed.spec.ts with bot-mode-row-click-mirrors-registry.spec.ts while our side had rewired its mock-server import. Kept upstream's replacement and rewired the three new specs importing ./mock-server to the consolidated tests-js copy (symbols verified present).
This commit is contained in:
yoniebans
2026-09-02 17:53:49 +02:00
841 changed files with 70213 additions and 4377 deletions
@@ -48,6 +48,9 @@ outputs:
installer:
description: Run the PowerShell installer tests on a Windows runner.
value: ${{ steps.classify.outputs.installer }}
desktop_updater:
description: Run the Windows desktop-update hand-off (windows.ps1) integration tests.
value: ${{ steps.classify.outputs.desktop_updater }}
rust:
description: Run `cargo test` for the Tauri bootstrap installer.
value: ${{ steps.classify.outputs.rust }}
@@ -0,0 +1,33 @@
name: Case Collision Check
# Rejects PRs that track two files whose paths differ only by case
# (README.md vs readme.md, src/Foo.py vs SRC/foo.py).
#
# Linux is case-sensitive; Windows and macOS (default) are not. A
# case-colliding pair lives fine in a Linux checkout and silently breaks
# every clone on a case-insensitive host — the filesystem can hold only
# one of them, so checkout fails or whichever wins overwrites the other.
# Git won't prevent the pair from landing (it only warns at checkout time,
# on a case-insensitive FS, for the client doing the checkout), so the only
# enforcement point is CI, on Linux, against the index.
#
# Runs unconditionally (no change-classifier gate): a collision can ship in
# any kind of PR — docs, JS, config, not just Python — so gating on a
# language lane would be the same "passive rule that cannot enforce a
# policy" trap the infographic check exists to close.
on:
workflow_call:
permissions:
contents: read
jobs:
check-case-collisions:
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Run case-collision checker
run: python3 scripts/check-case-collisions.py
+20 -8
View File
@@ -49,6 +49,7 @@ jobs:
uv_lock: ${{ steps.classify.outputs.uv_lock }}
npm_lock: ${{ steps.classify.outputs.npm_lock }}
installer: ${{ steps.classify.outputs.installer }}
desktop_updater: ${{ steps.classify.outputs.desktop_updater }}
rust: ${{ steps.classify.outputs.rust }}
docker_meta: ${{ steps.classify.outputs.docker_meta }}
mcp_catalog: ${{ steps.classify.outputs.mcp_catalog }}
@@ -84,6 +85,11 @@ jobs:
needs: detect
if: needs.detect.outputs.python == 'true'
uses: ./.github/workflows/tests-os.yml
with:
# The Windows lane spawns the real desktop-update hand-off script
# (tests/test_desktop_update_windows_*.py) only when that surface
# changed; unit-level windows_only tests always run.
desktop_updater: ${{ needs.detect.outputs.desktop_updater == 'true' }}
lint:
name: Python lints
@@ -122,14 +128,14 @@ jobs:
# Tests-only PRs (~17% of commits) skip this 5-minute job — the longest
# single job in the workflow — while still running the full pytest lanes.
#
# ⛔ TEMPORARILY DISABLED (Aug 2, 2026, Teknium) — the suite is red on
# every PR and on main itself since the Aug 1 night engines/npm churn
# (#76499 → #76562 → #76575): the mock-backend Electron window never
# gets a title, so boot/chat/setup/interim specs all fail identically
# regardless of the PR's diff (verified on #76573 and the docs-only
# #76582). Tracking issue: #76627 (assigned: Ari). To re-enable,
# delete the `false &&` below — nothing else changed.
if: ${{ false && (needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true') }}
# Re-disabled (Sep 2026): the Sep 1 re-enable is still incredibly flaky.
# Keep this a bare `if: false`. The earlier
# `${{ false && (... || ...) }}` form on this reusable-workflow job made
# GitHub's workflow parser fail at startup ("An unexpected error has
# occurred") — every ci.yaml run repo-wide dispatched 0 jobs from
# 24f5a60ed1 until this line changed. To re-enable, restore:
# if: ${{ needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' }}
if: false
uses: ./.github/workflows/e2e-desktop.yml
docs-site:
@@ -171,6 +177,11 @@ jobs:
needs: detect
uses: ./.github/workflows/profile-artifact-check.yml
case-collision-check:
name: Check no case-colliding filenames
needs: detect
uses: ./.github/workflows/case-collision-check.yml
lockfile-diff:
name: package-lock.json diff
needs: detect
@@ -233,6 +244,7 @@ jobs:
- history-check
- contributor-check
- uv-lockfile
- case-collision-check
- lockfile-diff
- docker-lint
- profile-artifact-check
+27
View File
@@ -27,6 +27,19 @@ name: OS-specific tests
on:
workflow_call:
inputs:
desktop_updater:
description: >-
Run the Windows desktop-update hand-off integration tests
(tests/test_desktop_update_windows_*.py). These spawn the real
scripts/desktop-update/windows.ps1 and poll its loopback server, so
they carry process-timing noise a shared runner amplifies; the
caller gates them on the classifier's desktop_updater lane so a PR
that never touched that surface cannot be failed by it. Push /
dispatch runs fail open (classifier sets every lane true).
type: boolean
required: false
default: true
permissions:
contents: read
@@ -134,9 +147,23 @@ jobs:
# would therefore abort the script on any non-zero exit and the
# exit-5 branch below would be unreachable dead code — the job
# would still fail red, but the diagnostic would never print.
# Desktop-update hand-off integration tests spawn the real
# windows.ps1; deselect them unless the PR touched that surface
# (see the workflow_call input). ``--ignore-glob`` keeps the file
# list above intact, so a renamed test file still trips the
# zero-tests guard rather than silently vanishing.
# (bash 3.2 on the macOS runner: an empty array under ``set -u`` is
# an unbound-variable error, hence the ``${arr[@]+...}`` idiom.)
EXTRA_ARGS=()
if [ "${{ inputs.desktop_updater }}" != "true" ]; then
echo "desktop_updater lane off: skipping tests/test_desktop_update_windows_*.py"
EXTRA_ARGS+=(--ignore-glob='*test_desktop_update_windows_*.py')
fi
status=0
uv run --no-sync python -m pytest \
"$@" \
${EXTRA_ARGS[@]+"${EXTRA_ARGS[@]}"} \
-m "${{ matrix.marker }} and not integration" \
-v --tb=short || status=$?
if [ "$status" -eq 5 ]; then
+3
View File
@@ -109,6 +109,9 @@ apps/shared/src/**/*.js
apps/shared/src/**/*.js.map
apps/shared/src/**/*.d.ts
apps/desktop/release/
# stage-and-swap Desktop rebuild output (#86443); removed after the swap, but
# a killed build must not leave the checkout dirty
apps/desktop/.staging-*/
*.tsbuildinfo
# Web UI assets — synced from @nous-research/ui at build time via
+37 -2
View File
@@ -1831,6 +1831,9 @@ def init_agent(
except Exception:
agent.show_commentary = True
# Window (seconds) for the bounded /fast auto|cold modes (agent.fast_mode).
agent.fast_auto_seconds = (_agent_cfg.get("agent") or {}).get("fast_auto_seconds", 60)
# LM Studio can either be explicitly preloaded through LM Studio's
# management API (the historical Hermes behavior) or left to LM Studio's
# just-in-time / Auto-Evict chat-completions path. Keep the default
@@ -1851,10 +1854,39 @@ def init_agent(
except Exception:
agent.lmstudio_load_mode = "explicit"
# API-transport streaming (``model.streaming``, default true). The
# conversation loop prefers ``stream=True`` for every turn — including
# subagent turns — to get fine-grained liveness health-checking (#3120),
# but self-hosted OpenAI-compatible backends with broken streaming
# tool-call paths (e.g. vLLM ``--tool-call-parser qwen3_xml`` + a
# reasoning parser can leak tool-call markup into plain text and return
# zero ``tool_calls``, #72901) silently no-op instead of executing.
# ``model.streaming: false`` seeds ``_disable_streaming`` so the session
# uses the non-streaming path, which the loop already falls back to at
# runtime when a provider rejects streaming. The setting is
# session-scoped: it persists across mid-session model switches, mirroring
# the runtime fallback's semantics. Orthogonal to ``display.streaming``
# (token rendering) — display-only settings are untouched.
agent._disable_streaming = False
try:
_model_section = _agent_cfg.get("model", {})
if isinstance(_model_section, dict):
_streaming = str(_model_section.get("streaming", "true")).strip().lower()
if _streaming in {"false", "0", "no", "off"}:
agent._disable_streaming = True
elif _streaming not in {"true", "1", "yes", "on"}:
logger.warning(
"Invalid model.streaming=%r; expected a boolean. Using streaming (default).",
_model_section.get("streaming"),
)
except Exception:
agent._disable_streaming = False
try:
agent._tool_guardrails = ToolCallGuardrailController(
ToolCallGuardrailConfig.from_mapping(
_agent_cfg.get("tool_loop_guardrails", {})
_agent_cfg.get("tool_loop_guardrails", {}),
platform=platform,
)
)
except Exception as _tlg_err:
@@ -2999,8 +3031,10 @@ def init_agent(
except Exception as _ce_err:
_ra().logger.debug("Context engine on_session_start: %s", _ce_err)
from agent.runtime_cwd import scope_terminal_cwd as _scope_terminal_cwd
agent._subdirectory_hints = SubdirectoryHintTracker(
working_dir=os.getenv("TERMINAL_CWD") or None,
working_dir=_scope_terminal_cwd() or None,
)
agent._user_turn_count = 0
# Copilot x-initiator flag: first API call of a user turn sends "user" (#3040).
@@ -3011,6 +3045,7 @@ def init_agent(
# until the first response with usage; invalidated on compaction and
# session switches so stale anchors can never suppress compression.
agent._usage_anchor = None
agent._turn_base_usage_anchor = None
# Cumulative token usage for the session
agent.session_prompt_tokens = 0
+3 -3
View File
@@ -108,7 +108,7 @@ def _ra():
AGENT_RUNTIME_POST_HOOK_TOOL_NAMES = frozenset(
{"todo", "session_search", "memory", "clarify", "read_terminal", "desktop_preview", "drive_preview", "annotate_preview", "read_window_below", "setup_mcp", "tour", "delegate_task"}
{"todo_list", "session_search", "memory", "clarify", "read_terminal", "desktop_preview", "drive_preview", "annotate_preview", "read_window_below", "setup_mcp", "gui_tour", "delegate_task"}
)
@@ -3529,7 +3529,7 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i
pass
return result
if function_name == "todo":
if function_name == "todo_list":
def _execute(next_args: dict) -> Any:
from tools.todo_tool import todo_tool as _todo_tool
return _finish_agent_tool(
@@ -3671,7 +3671,7 @@ def invoke_tool(agent, function_name: str, function_args: dict, effective_task_i
),
next_args,
)
elif function_name == "tour":
elif function_name == "gui_tour":
def _execute(next_args: dict) -> Any:
from tools.tour_tool import tour_tool as _tour_tool
return _finish_agent_tool(
+40 -13
View File
@@ -226,7 +226,7 @@ def _is_claude_model(model: str | None) -> bool:
return "claude" in (model or "").lower()
_FAST_MODE_SUPPORTED_SUBSTRINGS = ("opus-4-6", "opus-4.6")
_FAST_MODE_SUPPORTED_SUBSTRINGS = ("opus-4-8", "opus-4.8", "opus-5")
# ── Max output token limits per Anthropic model ───────────────────────
# Source: Anthropic docs + Cline model catalog. Anthropic's API requires
@@ -433,13 +433,28 @@ def _forbids_sampling_params(model: str) -> bool:
def _supports_fast_mode(model: str) -> bool:
"""Return True for models that support Anthropic Fast Mode (speed=fast).
"""Return True for models that accept the ``speed: "fast"`` request param.
Per Anthropic docs, fast mode is currently supported on Opus 4.6 only.
Sending ``speed: "fast"`` to any other Claude model (including Opus 4.7)
returns HTTP 400. This guard prevents silently 400'ing when stale config
or older callers leave fast mode enabled across a model upgrade.
Per the Anthropic fast-mode docs (research preview), the ``speed`` param
is supported on Opus 4.8 and Opus 5 — Claude API only. The matrix has
changed with nearly every Opus release, in both directions:
- Opus 4.6 HAD fast mode at launch and LOST it (2026-06-29): requests
with ``speed: "fast"`` do not error — they silently run at standard
speed and bill standard rates (``usage.speed: "standard"``). Keeping
4.6 in this allowlist would show users a fast toggle that does
nothing.
- Opus 4.7 never had it and hard-400s on the parameter.
- Dedicated ``…-fast`` model ids (e.g. OpenRouter's
``claude-opus-4.8-fast``) select fast inference via the model field
itself and must NOT also receive the speed parameter.
Keep this an explicit allowlist rather than a version-floor check so a
model that drops fast mode again fails closed (standard speed) instead
of silently 400'ing.
"""
if "-fast" in model:
return False
return any(v in model for v in _FAST_MODE_SUPPORTED_SUBSTRINGS)
@@ -935,9 +950,9 @@ def build_anthropic_kwargs(
thinking block signatures are stripped (they are Anthropic-proprietary).
When *fast_mode* is True, adds ``extra_body["speed"] = "fast"`` and the
fast-mode beta header for ~2.5x faster output throughput on Opus 4.6.
Currently only supported on native Anthropic endpoints (not third-party
compatible ones).
fast-mode beta header for ~2.5x faster output throughput on Opus 4.8 /
Opus 5. Currently only supported on native Anthropic endpoints (not
third-party compatible ones).
"""
system, anthropic_messages = convert_messages_to_anthropic(
messages, base_url=base_url, model=model
@@ -1148,12 +1163,15 @@ def build_anthropic_kwargs(
for _sampling_key in ("temperature", "top_p", "top_k"):
kwargs.pop(_sampling_key, None)
# ── Fast mode (Opus 4.6 only) ────────────────────────────────────
# ── Fast mode (Opus 4.8 / Opus 5) ────────────────────────────────
# Adds extra_body.speed="fast" + the fast-mode beta header for ~2.5x
# output speed. Per Anthropic docs, fast mode is only supported on
# Opus 4.6 — Opus 4.7 and other models 400 on the speed parameter.
# output speed. Per Anthropic docs the speed param is supported on
# Opus 4.8 and Opus 5 (research preview); Opus 4.7 400s on it and
# Opus 4.6 silently ignores it (standard speed, standard billing).
# Only for native Anthropic endpoints — third-party providers would
# reject the unknown beta header and speed parameter.
# reject the unknown beta header and speed parameter, and Anthropic
# itself scopes fast mode to the Claude API (not Bedrock/Vertex/
# Foundry).
if (
fast_mode
and not _is_third_party_anthropic_endpoint(base_url)
@@ -1282,12 +1300,21 @@ def create_anthropic_message(
for _event in stream:
try:
on_stream_event(_event)
except TimeoutError:
# The callback is the caller's deadline seam
# (#99692: the host waiting on this summary has
# already given up). Abandon the stream — the
# ``with`` closes it — instead of streaming an
# answer nobody will read.
raise
except Exception:
logger.debug(
"%son_stream_event callback failed",
log_prefix, exc_info=True,
)
return stream.get_final_message()
except TimeoutError:
raise
except Exception as exc:
if not _is_stream_unavailable_error(exc):
raise
+24 -1
View File
@@ -917,6 +917,22 @@ def _get_hermes_oauth_file() -> Path:
return get_hermes_home() / ".anthropic_oauth.json"
def _root_hermes_oauth_file() -> Optional[Path]:
"""Global-root ``.anthropic_oauth.json`` when running inside a named profile.
``None`` in classic mode (profile == root). Used to commit a rotation of a
grant the profile borrowed through the credential-pool root fallback.
"""
try:
from hermes_constants import get_default_hermes_root
root = get_default_hermes_root()
if root.resolve(strict=False) == get_hermes_home().resolve(strict=False):
return None
return root / ".anthropic_oauth.json"
except Exception:
return None
def _generate_pkce() -> tuple:
"""Generate PKCE code_verifier and code_challenge (S256)."""
import base64
@@ -1077,9 +1093,16 @@ def _write_hermes_oauth_credentials(
access_token: str,
refresh_token: Optional[str],
expires_at_ms: Optional[int],
*,
target: Optional[Path] = None,
) -> None:
"""Write refreshed hermes_pkce tokens back to ~/.hermes/.anthropic_oauth.json.
``target`` overrides the destination: a named profile that rotated a grant
it BORROWED from the global root (credential-pool root fallback) must
commit the new pair to the ROOT singleton, not create a forked copy under
its own HERMES_HOME (#100339).
Without this, a successful pool-level refresh of a ``hermes_pkce``-sourced
entry is invisible to this singleton file. The next ``load_pool()`` call
runs ``_seed_from_singletons()``, which reads the stale file and
@@ -1090,7 +1113,7 @@ def _write_hermes_oauth_credentials(
file, for the same reason ``_write_claude_code_credentials`` does: this is
the commit step of the refresh transaction.
"""
oauth_file = _get_hermes_oauth_file()
oauth_file = target if target is not None else _get_hermes_oauth_file()
try:
oauth_data = {
"accessToken": access_token,
+251 -21
View File
@@ -453,6 +453,16 @@ _aux_progress = threading.local()
_aux_dispatch = threading.local()
_aux_provider_response = threading.local()
# Absolute wall-clock deadline (time.monotonic) of the HOST waiting for this
# auxiliary call, when it has one (#99692). Liveness alone is not enough: a
# host also stops waiting at its own total ceiling, and the streamed consumer
# below bounds itself only by _aux_stream_total_ceiling() — a budget derived
# from the aux request timeout, which is >= the host ceiling for every
# configured value AND starts counting later. So the stream that outlives its
# abandoned host is not an edge case; it is the guaranteed outcome of every
# total-ceiling timeout.
_aux_stream_deadline = threading.local()
def _notify_aux_progress() -> None:
"""Tick the installed forward-progress hook, if any. Never raises."""
@@ -525,6 +535,37 @@ def _anthropic_event_has_content(event: Any) -> bool:
return False
def _anthropic_aux_stream_event_hook() -> Callable[[Any], None]:
"""Per-event callback for the Anthropic auxiliary wire.
Records provider-response timing for every frame, ticks the forward-progress
hook only for substantive payloads (keepalive pings must not keep a stalled
summary alive), and — #99692 — stops the stream at the waiting host's
absolute deadline (``aux_stream_deadline``) or on an explicit hard cancel,
the same two stop conditions the chat.completions and Codex wires honour.
The ``TimeoutError`` is phrased with "timed out" so ``_is_timeout_error``
classifies it like any other request timeout.
"""
host_deadline = _current_aux_stream_deadline()
started = time.monotonic()
def _on_event(event: Any) -> None:
if _anthropic_event_has_content(event):
_notify_aux_provider_response()
else:
_notify_aux_timing_response()
if _aux_interrupt_cancel_requested():
raise AuxiliaryExplicitCancellation()
if host_deadline is not None and time.monotonic() >= host_deadline:
raise TimeoutError(
"Anthropic auxiliary stream timed out at the host compression "
f"deadline after {time.monotonic() - started:.0f}s "
"(the caller already stopped waiting)"
)
return _on_event
_CODEX_PROGRESS_DELTA_TYPES = frozenset(
{
"response.output_text.delta",
@@ -587,6 +628,38 @@ def aux_progress_hook(hook):
yield
def _current_aux_stream_deadline() -> Optional[float]:
"""The waiting host's absolute monotonic deadline, if one is installed."""
return getattr(_aux_stream_deadline, "value", None)
@contextlib.contextmanager
def aux_stream_deadline(deadline: Optional[float]):
"""Publish the waiting host's absolute deadline to the stream consumer.
*deadline* is a ``time.monotonic()`` timestamp — the same instant the host
itself stops waiting — or ``None`` for callers with no host deadline (a
no-op passthrough, so callers can wire it unconditionally). Re-entrant-safe.
#99692: the progress hook is a one-way channel (worker -> host). This is the
return leg. ``8207862212`` releases the compression OWNER when the fence is
cancelled, but the isolated provider daemon
(:func:`_run_protected_sync_provider_call`) that holds the socket keeps
streaming to its own ``_aux_stream_total_ceiling`` budget — >= the host's
ceiling by construction — billing an abandoned summary the commit fence is
already guaranteed to refuse, and stacking one fresh orphan per turn on a
session that compression never managed to shrink.
"""
previous = getattr(_aux_stream_deadline, "value", None)
_aux_stream_deadline.value = (
deadline if isinstance(deadline, (int, float)) else previous
)
try:
yield
finally:
_aux_stream_deadline.value = previous
# Back-compat alias — the timing hooks were introduced with this name.
_aux_timing_hook = _aux_thread_local_hook
@@ -629,6 +702,11 @@ def _run_protected_sync_provider_call(
# the protected daemon path is taken.
dispatch_hook = getattr(_aux_dispatch, "hook", None)
provider_response_hook = getattr(_aux_provider_response, "hook", None)
# #99692: the stream is consumed on the daemon below, and thread-locals do
# not cross that boundary — an owner-thread-only deadline would leave the
# fix inert on exactly the path large-session compression takes (protected
# call + hard-cancel source installed).
host_deadline = _current_aux_stream_deadline()
provider_context = contextvars.copy_context()
done = threading.Event()
outcome: dict[str, Any] = {}
@@ -639,6 +717,7 @@ def _run_protected_sync_provider_call(
aux_progress_hook(progress_hook),
_aux_thread_local_hook(_aux_dispatch, dispatch_hook),
_aux_thread_local_hook(_aux_provider_response, provider_response_hook),
aux_stream_deadline(host_deadline),
aux_interrupt_protection(cancel_check=cancel_check),
):
outcome["result"] = callback(kwargs)
@@ -993,6 +1072,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
network path — the underlying fetch is memory+disk cached with a
last-known-good fallback.
"""
is_nous = provider_id.strip().lower() == "nous"
try:
from hermes_cli.auth import resolve_api_key_provider_credentials
from hermes_cli.models import fetch_models_with_pricing
@@ -1012,6 +1092,17 @@ def _fast_model_from_catalog(provider_id: str) -> str:
# fetch below still works for the catalogs that allow it.
logger.debug("No credentials for %s catalog", provider_id, exc_info=True)
if not api_key and is_nous:
# Nous is OAuth, so the resolver above raises for it. An anonymous
# read returns the full catalog, and a model picked from it is
# refused at request time by the org's policy.
try:
from hermes_cli.models import _resolve_nous_pricing_credentials
api_key, base_url = _resolve_nous_pricing_credentials()
except Exception:
logger.debug("No Nous credentials for catalog", exc_info=True)
if not base_url:
base_url = str(getattr(get_provider_profile(provider_id), "base_url", "") or "")
base_url = base_url.rstrip("/")
@@ -1020,14 +1111,37 @@ def _fast_model_from_catalog(provider_id: str) -> str:
# fetch_models_with_pricing appends its own /v1/models.
if base_url.endswith("/v1"):
base_url = base_url[:-3]
# Same entry the pickers use, so the Nous-only arguments must match
# theirs: seeding it here without them costs the picker its sale chrome
# and leaves the policy catalog with no expiry.
_nous_kwargs = {}
if is_nous:
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
_nous_kwargs = {
"include_sale_original": True,
"cache_ttl_seconds": _NOUS_CATALOG_TTL_SECONDS,
}
catalog = fetch_models_with_pricing(
api_key=api_key or None, base_url=base_url, timeout=3.0
api_key=api_key or None, base_url=base_url, timeout=3.0, **_nous_kwargs
) or {}
except Exception:
logger.debug("Fast-model catalog lookup failed for %s", provider_id, exc_info=True)
return ""
ids = sorted((str(m) for m in catalog), key=_model_recency_key, reverse=True)
if is_nous:
# The catalog's keys are a source of ids here, so the policy narrows
# them as it does the pickers' lists.
try:
from hermes_cli.models import (
nous_policy_allowed_ids,
restrict_to_nous_policy,
)
ids = restrict_to_nous_policy(ids, nous_policy_allowed_ids())
except Exception:
logger.debug("Nous policy filter unavailable", exc_info=True)
for family in _FAST_MODEL_FAMILIES:
for model_id in ids:
lowered = model_id.lower()
@@ -1036,6 +1150,18 @@ def _fast_model_from_catalog(provider_id: str) -> str:
return ""
def _nous_policy_blocks(model_id: str) -> bool:
"""True when the org's model policy does not admit *model_id*."""
try:
from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
allowed = nous_policy_allowed_ids()
return bool(allowed) and not restrict_to_nous_policy([model_id], allowed)
except Exception:
logger.debug("Nous policy check unavailable", exc_info=True)
return False
# Default auxiliary models for direct API-key providers (cheap/fast for side tasks)
def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False) -> str:
"""Return the cheap auxiliary model for a provider.
@@ -1063,21 +1189,26 @@ def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False)
except Exception:
pass
picked = ""
if prefer_fast:
catalog_pick = _fast_model_from_catalog(provider_id)
if catalog_pick:
return catalog_pick
if profile is not None:
picked = _fast_model_from_catalog(provider_id)
if not picked and profile is not None:
try:
live = profile.resolve_aux_model()
if live:
return live
picked = profile.resolve_aux_model() or ""
except Exception:
logger.debug("resolve_aux_model failed for %s", provider_id, exc_info=True)
if profile is not None and profile.default_aux_model:
return profile.default_aux_model
return _API_KEY_PROVIDER_AUX_MODELS_FALLBACK.get(provider_id, "")
if not picked and profile is not None and profile.default_aux_model:
picked = profile.default_aux_model
if not picked:
picked = _API_KEY_PROVIDER_AUX_MODELS_FALLBACK.get(provider_id, "")
# Steps 2-4 are policy-blind: resolve_aux_model queries a public
# recommendation and the rest are hardcoded. A blocked pick is refused at
# request time, so drop it and let the caller keep the main model.
if picked and provider_id.strip().lower() == "nous" and _nous_policy_blocks(picked):
return ""
return picked
@@ -1837,6 +1968,15 @@ class _CodexCompletionsAdapter:
if total_timeout is not None:
no_progress_timeout = min(no_progress_timeout, float(total_timeout))
hard_deadline = _start_monotonic + _aux_stream_total_ceiling(total_timeout)
# #99692: the waiting host's absolute deadline (compress_context
# publishes its commit-fence ceiling via aux_stream_deadline) clamps
# the hard ceiling so the re-armable watchdog Timer wakes and severs
# the socket at the instant the host stops waiting — a live Codex
# stream cannot otherwise be stopped by a per-event cancel check
# while it is blocked between events.
_host_deadline = _current_aux_stream_deadline()
if isinstance(_host_deadline, (int, float)) and _host_deadline < hard_deadline:
hard_deadline = float(_host_deadline)
deadline_lock = threading.Lock()
progress_deadline = [_start_monotonic + no_progress_timeout]
saw_content = threading.Event()
@@ -2452,13 +2592,7 @@ class _AnthropicCompletionsAdapter:
# stalled summary open. No-op when no hook is installed (None
# keeps the fast get_final_message path).
on_stream_event=(
(
lambda event: (
_notify_aux_provider_response()
if _anthropic_event_has_content(event)
else _notify_aux_timing_response()
)
)
_anthropic_aux_stream_event_hook()
if _aux_progress_active()
else None
),
@@ -9187,6 +9321,14 @@ def _build_call_kwargs(
_provider_norm == "openrouter"
or base_url_host_matches(_effective_base, "openrouter.ai")
)
# The managed local llama-server honors explicit caps too: a local
# decode burns the user's own GPU at full tilt, so a caller that
# says "this is a 64-token task" must be believed — an uncapped
# local generation whose EOS never comes runs to the full context
# window. No wire-format quirks apply (llama.cpp accepts
# max_tokens), and the no-default-cap policy is unchanged: this
# only forwards caps callers explicitly set.
_is_managed_local = _is_managed_local_endpoint(_effective_base)
if (
_is_anthropic_compat_endpoint(provider, _effective_base)
or _nous_on_messages
@@ -9194,6 +9336,7 @@ def _build_call_kwargs(
or _is_moa
or _is_gemini_native
or _is_openrouter
or _is_managed_local
):
# Use auxiliary_max_tokens_param() so models that require
# max_completion_tokens (GPT-5 family, Copilot) get the right
@@ -9539,6 +9682,49 @@ def _is_streaming_rejected_error(exc: Exception) -> bool:
)
_MANAGED_LOCAL_STATE_TTL_S = 15.0
_managed_local_cache: "tuple[float, str]" = (0.0, "")
def _managed_local_netloc() -> str:
"""host:port of the managed local llama-server, or "" when none.
Read from the supervisor's state file (written at spawn, removed on
stop) with a short TTL so per-request checks don't hit the disk. The
state file is the same source provider resolution uses, so the match
is exact — no false positives on other localhost endpoints.
"""
global _managed_local_cache
now = time.monotonic()
ts, cached = _managed_local_cache
if now - ts < _MANAGED_LOCAL_STATE_TTL_S:
return cached
netloc = ""
try:
from hermes_cli.local_runtime.supervisor import state_path
raw = state_path().read_text(encoding="utf-8")
base = str((json.loads(raw) or {}).get("base_url", ""))
netloc = urlparse(base).netloc.lower()
except Exception:
netloc = ""
_managed_local_cache = (now, netloc)
return netloc
def _is_managed_local_endpoint(base_url: Optional[str]) -> bool:
"""True when *base_url* targets the llama-server this Hermes manages."""
if not base_url:
return False
managed = _managed_local_netloc()
if not managed:
return False
try:
return urlparse(str(base_url)).netloc.lower() == managed
except Exception:
return False
def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
"""Detect providers that only accept streaming (non-stream = HTTP 400).
@@ -9554,6 +9740,18 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
Beyond the known-host list, users can mark ANY custom endpoint as
stream-only via ``auxiliary.stream_only_base_urls`` in config.yaml
(list of substrings matched against the endpoint URL).
The managed local llama-server is always streamed for a different
reason: cancellation. llama-server only notices a dead client when it
writes to the socket. A non-streamed request writes once — after the
FULL generation — so an abandoned call (client timeout, retry, app
exit) keeps the GPU decoding to the end of the context window with
nobody listening; requests that queue behind a model load are the
worst case, since the client is long gone before decode even starts.
Streaming writes every few tokens, so an abandoned decode dies at the
first post-disconnect chunk (verified against llama-server b10362:
streamed disconnect cancels in <1s through the router; non-streamed
survives until the server's next incidental socket poll, if ever).
"""
_url = str(base_url or "").lower()
if not _url:
@@ -9561,6 +9759,9 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
# Tencent Copilot — "Non-stream chat request is currently not supported"
if base_url_host_matches(_url, "copilot.tencent.com"):
return True
# Managed local llama-server — streamed so abandonment cancels decode.
if _is_managed_local_endpoint(_url):
return True
try:
from hermes_cli.config import load_config
aux_cfg = (load_config() or {}).get("auxiliary", {})
@@ -9753,7 +9954,11 @@ def _aggregate_chat_stream(
Accumulation is shared with the async mirror via
:class:`_ChatStreamAccumulator`.
"""
acc = _ChatStreamAccumulator(model=model, total_ceiling=total_ceiling)
acc = _ChatStreamAccumulator(
model=model,
total_ceiling=total_ceiling,
host_deadline=_current_aux_stream_deadline(),
)
try:
for chunk in chunks:
acc.feed(chunk)
@@ -9775,9 +9980,20 @@ class _ChatStreamAccumulator:
tool-call delta reassembly, same "timed out" ceiling phrasing).
"""
def __init__(self, model: str = "", total_ceiling: Optional[float] = None):
def __init__(
self,
model: str = "",
total_ceiling: Optional[float] = None,
host_deadline: Optional[float] = None,
):
self._started = time.monotonic()
self._total_ceiling = total_ceiling
# #99692: absolute instant the WAITING HOST gives up. Checked as well
# as (not instead of) the ceiling above: the ceiling still bounds
# callers with no host deadline, and the host deadline is absolute, so
# it is unaffected by however long dispatch and TTFT took before this
# accumulator was constructed.
self._host_deadline = host_deadline
self.content_parts: List[str] = []
self.reasoning_parts: List[str] = []
self.reasoning_details: List[Any] = []
@@ -9801,6 +10017,16 @@ class _ChatStreamAccumulator:
f"Auxiliary streamed call timed out after {self._total_ceiling:.0f}s "
"total ceiling (stream still open but over budget)"
)
if (
self._host_deadline is not None
and time.monotonic() >= self._host_deadline
):
raise TimeoutError(
"Auxiliary streamed call timed out at the host compression "
f"deadline after {time.monotonic() - self._started:.0f}s "
"(the caller already stopped waiting; streaming on would only "
"pin its session lease)"
)
self.resp_id = getattr(chunk, "id", None) or self.resp_id
self.resp_model = getattr(chunk, "model", None) or self.resp_model
chunk_usage = getattr(chunk, "usage", None)
@@ -9912,7 +10138,11 @@ async def _aggregate_chat_stream_async(
the sync helper raises. Same accumulation and ceiling semantics via
:class:`_ChatStreamAccumulator`.
"""
acc = _ChatStreamAccumulator(model=model, total_ceiling=total_ceiling)
acc = _ChatStreamAccumulator(
model=model,
total_ceiling=total_ceiling,
host_deadline=_current_aux_stream_deadline(),
)
try:
async for chunk in chunks:
acc.feed(chunk)
+432 -149
View File
@@ -34,6 +34,7 @@ from agent.error_classifier import (
PROVIDER_STREAM_NON_JSON_ERROR_CODE,
)
from agent.errors import EmptyStreamError
from agent.fast_mode import effective_request_overrides
from agent.turn_context import substitute_api_content
from agent.gemini_native_adapter import is_native_gemini_base_url
from agent.model_metadata import is_local_endpoint
@@ -1055,6 +1056,59 @@ def should_use_direct_api_call(agent) -> bool:
_DIRECT_API_ACTIVITY_HEARTBEAT_SECONDS = 15.0
def _managed_local_load_notice(agent, api_kwargs: dict) -> "Optional[str]":
"""A live phase notice while the managed local server works before the
first token, or None when neither phase (nor the managed server) applies:
- "⏳ loading <model> into memory — N%" (weights streaming off disk;
real per-tensor percent from the router's SSE stream)
- "⚙ processing prompt — N of ~M tokens (P%)" (prefill; live counter
from /slots, denominator estimated from the request body)
A cold local model spends ~tens of seconds loading and a long-context
turn spends tens more in prefill; without this, both windows render as
the generic "no output yet (provider may be slow or overloaded)" stall
warning — alarming copy for healthy, expected phases.
"""
try:
base = str(getattr(agent, "base_url", "") or "")
if not base:
return None
import json as _json
from urllib.parse import urlparse
from hermes_cli.local_runtime.load_progress import (
get_loading_progress,
get_prefill_progress,
)
from hermes_cli.local_runtime.supervisor import state_path
state = _json.loads(state_path().read_text(encoding="utf-8"))
managed = urlparse(str(state.get("base_url", ""))).netloc.lower()
if not managed or urlparse(base).netloc.lower() != managed:
return None
model = str(api_kwargs.get("model", ""))
progress = get_loading_progress().get(model)
if progress is not None:
return (
f"⏳ loading {model} into memory — {progress['percent']}% "
"(responses start once the model is loaded)"
)
prefill = get_prefill_progress(model)
if prefill is not None:
processed = int(prefill["processed"])
total = estimate_request_context_tokens(api_kwargs)
if total and total >= processed:
pct = max(0, min(100, round(processed / total * 100)))
return f"⚙ processing prompt — {pct}%"
# Counter past the estimate (estimator undercounted): no honest
# denominator, so no percent — the UI shows label-only.
return "⚙ processing prompt"
return None
except Exception: # noqa: BLE001 — a status nicety must never break a call
return None
def _resolve_direct_stale_timeout(agent, api_kwargs: dict) -> float:
"""Stale budget for the inline non-streaming call.
@@ -1379,6 +1433,35 @@ def interruptible_api_call(agent, api_kwargs: dict):
# a network bug and surfaced to the caller. (PR #6600 — cascading interrupt
# hang.)
_request_cancelled = {"value": False}
# Codex Responses retirement token (codex_responses only). The worker
# thread reads it through ``agent._active_codex_stream_request_token`` to
# tell whether it still owns the turn. When a watchdog below force-closes
# the connection it clears the agent-level token, so a worker still
# draining SSE frames raises instead of returning its partial output as a
# "completed" response (see run_codex_stream's _request_is_current).
# ``_codex_request_retired`` is the request-local mirror, used to swallow
# the transport error our own force-close causes — same split as
# ``_request_cancelled`` above.
_codex_request_token = object() if agent.api_mode == "codex_responses" else None
_codex_request_retired = {"value": False}
def _install_codex_request_token() -> None:
if _codex_request_token is None:
return
if _codex_request_retired["value"]:
# Already retired before the worker got going — do not re-publish.
return
agent._active_codex_stream_request_token = _codex_request_token
def _retire_codex_request_token() -> None:
if _codex_request_token is None:
return
_codex_request_retired["value"] = True
if (
getattr(agent, "_active_codex_stream_request_token", None)
is _codex_request_token
):
agent._active_codex_stream_request_token = None
def _set_request_client(client, *, kind: str = "openai"):
with request_client_lock:
@@ -1438,6 +1521,7 @@ def interruptible_api_call(agent, api_kwargs: dict):
def _call():
try:
_install_codex_request_token()
# _set_request_client registers each per-request client with the
# stranger-thread abort machinery above; the shared dispatch helper
# builds it via this callback (openai- or anthropic-kind) so the
@@ -1460,15 +1544,34 @@ def interruptible_api_call(agent, api_kwargs: dict):
# handler, the transport error is the expected consequence of our
# own force-close, NOT a network bug. Swallow it instead of
# surfacing — the main thread raises InterruptedError. (#6600)
if _request_cancelled["value"]:
logger.debug(
"Non-streaming worker caught %s after request cancellation — "
"exiting without surfacing a network error.",
type(e).__name__,
)
if _request_cancelled["value"] or _codex_request_retired["value"]:
# Retirement is logged at info: it means a watchdog discarded
# output the provider had already sent, which is exactly the
# event an operator debugging a truncated reply needs to see.
# Cancellation stays at debug — a user interrupt is a normal,
# high-frequency outcome and the caller already surfaces it.
if _codex_request_retired["value"]:
logger.info(
"Codex worker caught %s after request retirement — "
"discarding the stale partial instead of surfacing it "
"as a completed response. %s",
type(e).__name__,
agent._client_log_context(),
)
else:
logger.debug(
"Non-streaming worker caught %s after request "
"cancellation — exiting without surfacing a network "
"error.",
type(e).__name__,
)
return
result["error"] = e
finally:
# Retire first: _close_request_client_once can raise (every other
# call site wraps it in try/except), and a leaked token would let a
# later worker mistake itself for the owning attempt.
_retire_codex_request_token()
# Reuse reason only on a clean response; any other outcome —
# error, or the cancel-swallow return above (which leaves both
# result slots None) — really closes so the next attempt builds
@@ -1681,6 +1784,7 @@ def interruptible_api_call(agent, api_kwargs: dict):
_close_request_client_once("codex_ttfb_kill")
except Exception:
pass
_retire_codex_request_token()
agent._emit_wait_notice(
f"⚠ no response from provider in {int(_elapsed)}s — "
f"reconnecting..."
@@ -1731,6 +1835,7 @@ def interruptible_api_call(agent, api_kwargs: dict):
_close_request_client_once("codex_stream_idle_kill")
except Exception:
pass
_retire_codex_request_token()
agent._touch_activity(
f"codex stream killed after {int(_event_stale_elapsed)}s with no SSE events"
)
@@ -1762,6 +1867,7 @@ def interruptible_api_call(agent, api_kwargs: dict):
_close_request_client_once("stale_call_kill")
except Exception:
pass
_retire_codex_request_token()
# Circuit breaker (#58962): count the stale kill. See the
# canonical comment block above ``_stale_streak()``.
_bump_stale_streak(agent)
@@ -1809,6 +1915,7 @@ def interruptible_api_call(agent, api_kwargs: dict):
_close_request_client_once("interrupt_abort")
except Exception:
pass
_retire_codex_request_token()
# #81521 (sibling of the streaming-path fix): wait for the worker
# to unwind Relay-managed scopes before surfacing
# InterruptedError, so turn teardown cannot race a still-open
@@ -1826,10 +1933,63 @@ def interruptible_api_call(agent, api_kwargs: dict):
def _consume_ephemeral_reasoning_off(agent) -> bool:
"""Consume the one-shot "answer without thinking" continuation flag.
Set by the length-continuation path when a request returned reasoning
but NO visible content — the thinking phase consumed the entire output
cap (GLM-5.3 on ollama-cloud with reasoning_effort=high: reported live as
finish_reason="length", content="", completion_tokens == max_tokens).
Continuation turns never replay the prior reasoning, so re-running with
thinking ON re-derives — and re-burns — the whole thinking budget from
scratch instead of writing the answer (observed: 4 futile continuations
then "Response remained truncated after 4 continuation attempts").
When True is returned the caller must override the wire reasoning_config
with ``{"enabled": False, "effort": "none"}`` for exactly the next call.
Prompt-cache cost (deliberate, bounded): the reasoning parameter is part
of the provider's cache key on config-sensitive providers — Anthropic
renders thinking/effort into the prompt, OpenAI lists reasoning.effort
among prefix-affecting settings — so THAT one request misses the prefix
cache and pays a cold write of the full prefix (1.25x input instead of
the 0.1x read). The next request goes out with the configured reasoning
again and hits the thinking-on entry written by the truncated request
(still within TTL), so the damage is exactly one write. Template-tail
providers (GLM/Qwen/Kimi-style, where thinking on/off is a chat-template
switch at the tail) see no prefix change at all. The system prompt bytes
are never touched. This is far cheaper than what the flag prevents: four
full-output-budget requests that produce nothing and end the turn with an
error.
"""
if getattr(agent, "_ephemeral_reasoning_off", False):
agent._ephemeral_reasoning_off = False
return True
return False
def _reasoning_config_for_wire(agent):
"""``agent.reasoning_config`` with the one-shot reasoning-off override applied."""
if _consume_ephemeral_reasoning_off(agent):
return {
**(agent.reasoning_config or {}),
"enabled": False,
"effort": "none",
}
return agent.reasoning_config
def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = None) -> dict:
"""Build the keyword arguments dict for the active API mode."""
# One-shot continuation override — consumed exactly once, on the FIRST
# request this call builds (only one api_mode branch runs per invocation).
_wire_reasoning_config = _reasoning_config_for_wire(agent)
if tools_for_api is None:
tools_for_api = agent.tools
# The one place request_overrides are consumed: static /fast values are
# already pinned in agent.request_overrides; auto/cold windows layer the
# fast override here, per request, only while the window is open.
_request_overrides = effective_request_overrides(agent)
if agent.api_mode == "anthropic_messages":
_transport = agent._get_transport()
@@ -1844,12 +2004,12 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
messages=anthropic_messages,
tools=tools_for_api,
max_tokens=ephemeral_out if ephemeral_out is not None else agent.max_tokens,
reasoning_config=agent.reasoning_config,
reasoning_config=_wire_reasoning_config,
is_oauth=agent._is_anthropic_oauth,
preserve_dots=agent._anthropic_preserve_dots(),
context_length=ctx_len,
base_url=getattr(agent, "_anthropic_base_url", None),
fast_mode=(agent.request_overrides or {}).get("speed") == "fast",
fast_mode=_request_overrides.get("speed") == "fast",
drop_context_1m_beta=bool(getattr(agent, "_oauth_1m_beta_disabled", False)),
)
# Nous Portal reads ``tags`` and ``session_id`` as top-level body fields
@@ -1936,13 +2096,13 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
model=agent.model,
messages=_msgs_for_codex,
tools=tools_for_api,
reasoning_config=agent.reasoning_config,
reasoning_config=_wire_reasoning_config,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_cache_scope_id,
base_url=agent.base_url,
max_tokens=agent.max_tokens,
timeout=agent._resolved_api_call_timeout(),
request_overrides=agent.request_overrides,
request_overrides=_request_overrides,
provider=getattr(agent, "provider", None),
is_github_responses=is_github_responses,
is_codex_backend=is_codex_backend,
@@ -2093,8 +2253,8 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
max_tokens=agent.max_tokens,
ephemeral_max_output_tokens=_ephemeral_out,
max_tokens_param_fn=agent._max_tokens_param,
reasoning_config=agent.reasoning_config,
request_overrides=agent.request_overrides,
reasoning_config=_wire_reasoning_config,
request_overrides=_request_overrides,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_cache_scope_id,
provider_profile=_profile,
@@ -2126,8 +2286,8 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
max_tokens=agent.max_tokens,
ephemeral_max_output_tokens=_ephemeral_out,
max_tokens_param_fn=agent._max_tokens_param,
reasoning_config=agent.reasoning_config,
request_overrides=agent.request_overrides,
reasoning_config=_wire_reasoning_config,
request_overrides=_request_overrides,
session_id=getattr(agent, "session_id", None),
cache_scope_id=_cache_scope_id,
model_lower=(agent.model or "").lower(),
@@ -3493,14 +3653,11 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
if emit is not None:
emit(final_text=final_text, finished=finished, error=error)
# Cron and other non-interactive, nested-pool contexts deadlock on the
# spawned worker thread (#62151). They also have no stream consumer, so the
# deltas this path produces go nowhere. Delegate to the non-streaming entry
# (which runs inline via should_use_direct_api_call) exactly like the codex
# branch below — routing through the _interruptible_api_call method keeps the
# outer loop's per-request retry/refresh seam intact.
if should_use_direct_api_call(agent):
return agent._interruptible_api_call(api_kwargs)
# Cron turns and delegated children (should_use_direct_api_call) used to be
# short-circuited here onto the NON-streaming wire. They now stay on this
# streaming path and run the request inline — see the ``_inline`` block
# before the poll loop below. Only the codex branch still detours through
# _interruptible_api_call (it streams internally).
if agent.api_mode == "codex_responses":
# Codex streams internally via _run_codex_stream. The main dispatch
@@ -4130,6 +4287,29 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
provider_tool_in_flight["yes"] = True
except Exception:
pass
# Payload-empty terminal chunk: the provider completed the
# stream (`finish_reason` set, no further writable delta). The
# attempt/writer fence exists to stop a superseded stream from
# writing *more* text. Fending this marker-only chunk discards
# the only completion signal, which the drop-guard then
# mislabels as a mid-stream drop. A finish chunk that still
# carries content/tool_calls remains gated.
try:
_choices = getattr(_chunk, "choices", None)
if _choices:
_choice = _choices[0]
if getattr(_choice, "finish_reason", None):
_delta = getattr(_choice, "delta", None)
_has_write = bool(
getattr(_delta, "content", None)
or getattr(_delta, "tool_calls", None)
or getattr(_delta, "reasoning_content", None)
or getattr(_delta, "reasoning", None)
)
if not _has_write:
return True
except Exception:
pass
if not _stream_attempt_is_active(stream_attempt_id):
return False
token = _writer_token["value"]
@@ -5295,143 +5475,246 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
if _reasoning_floor is not None:
_stream_stale_timeout = max(_stream_stale_timeout, _reasoning_floor)
t = threading.Thread(target=_context_thread_target(_call), daemon=True)
t.start()
# Delegated children and gateway cron turns run the streaming request
# INLINE on the conversation thread: spawning the interrupt worker inside
# their nested thread pools wedges before the socket opens (#62151,
# #60203). They used to be routed to the non-streaming wire for that
# reason — but streaming is also the transport keepalive and the
# liveness signal: a non-streaming POST that stays silent through a
# reasoning model's thinking phase is killed by edge proxies (z.ai 524,
# #90202) and by our own stale watchdog, which cannot tell thinking from
# a hang when no bytes ever arrive (#100260). Inline mode keeps the
# stream (per-token liveness) and moves ONLY the lightweight poll loop
# below — heartbeat, stale detector, interrupt abort — onto a monitor
# thread. The monitor never issues a request, so the no-worker property
# that fixes the deadlock class is preserved (same shape as the
# direct_api_call watchdog timer).
_inline = should_use_direct_api_call(agent)
_call_done = threading.Event()
_monitor_interrupted = {"yes": False}
def _run_call():
try:
_call()
finally:
_call_done.set()
if _inline:
t = None
else:
t = threading.Thread(target=_context_thread_target(_run_call), daemon=True)
t.start()
def _call_alive() -> bool:
return not _call_done.is_set()
def _wait_call(timeout: float) -> None:
_call_done.wait(timeout=timeout)
_last_heartbeat = time.time()
_HEARTBEAT_INTERVAL = 30.0 # seconds between gateway activity touches
while t.is_alive():
t.join(timeout=0.3)
# Managed local server: a cold model streams weights off disk for tens
# of seconds before the first token can exist. Surface THAT immediately
# (real per-tensor percent from the router's SSE stream) instead of
# letting the wait fall through to the 30s "provider may be slow or
# overloaded" copy. Checked on a ~1s cadence only while no chunks have
# arrived; the probe is an in-memory snapshot read, not a network call.
_last_load_poll = 0.0
_load_notice_shown = False
_load_notice_misses = 0
_is_local_base = bool(agent.base_url) and is_local_endpoint(agent.base_url)
# Periodic heartbeat: touch the agent's activity tracker so the
# gateway's inactivity monitor knows we're alive while waiting
# for stream chunks. Without this, long thinking pauses (e.g.
# reasoning models) or slow prefill on local providers (Ollama)
# trigger false inactivity timeouts. The _call thread touches
# activity on each chunk, but the gap between API call start
# and first chunk can exceed the gateway timeout — especially
# when the stale-stream timeout is disabled (local providers).
_hb_now = time.time()
if _hb_now - _last_heartbeat >= _HEARTBEAT_INTERVAL:
_last_heartbeat = _hb_now
_waiting_secs = int(_hb_now - last_chunk_time["t"])
if _waiting_secs >= _HEARTBEAT_INTERVAL:
# No chunks for 30s+ — rewrite the live spinner/status line
# so CLI/TUI/Desktop users see WHAT the wait is (slow or
# overloaded provider / long thinking pause) instead of an
# unexplained generic spinner, and WHEN recovery kicks in.
if (
_stream_stale_timeout is not None
and _stream_stale_timeout != float("inf")
):
_recovery = f"; auto-reconnect at {int(_stream_stale_timeout)}s"
def _monitor_loop() -> None:
nonlocal _last_heartbeat, _last_load_poll, _load_notice_shown, _load_notice_misses
while _call_alive():
_wait_call(0.3)
_hb_now = time.time()
# Cold-load window: last_chunk_time is touched at request-client
# creation and then only by REAL chunks, so "no chunk for 2s+" is
# true through a model load (nothing can stream while the child is
# still mapping weights) and false during healthy token flow —
# which is what keeps this poll off the streaming hot path. The
# probe itself is an in-memory snapshot read.
if (
_is_local_base
and _hb_now - last_chunk_time["t"] >= 2.0
and _hb_now - _last_load_poll >= 1.0
):
_last_load_poll = _hb_now
_load_notice = _managed_local_load_notice(agent, api_kwargs)
if _load_notice is not None:
agent._emit_wait_notice(_load_notice)
agent._touch_activity("local model loading")
_load_notice_shown = True
_load_notice_misses = 0
# Loading IS liveness for the heartbeat; the stale detector
# needs no help — the local floor (900s) dwarfs any load.
_last_heartbeat = _hb_now
continue
if _load_notice_shown:
# One missed sample is routine (a /slots read straddling a
# batch boundary, a 2s probe timeout under load) — clearing
# on it made the status line strobe blank once every few
# seconds mid-prefill. Only a SUSTAINED absence means the
# phase really ended.
_load_notice_misses += 1
if _load_notice_misses >= 3:
_load_notice_shown = False
_load_notice_misses = 0
agent._emit_wait_notice("")
# Periodic heartbeat: touch the agent's activity tracker so the
# gateway's inactivity monitor knows we're alive while waiting
# for stream chunks. Without this, long thinking pauses (e.g.
# reasoning models) or slow prefill on local providers (Ollama)
# trigger false inactivity timeouts. The _call thread touches
# activity on each chunk, but the gap between API call start
# and first chunk can exceed the gateway timeout — especially
# when the stale-stream timeout is disabled (local providers).
if _hb_now - _last_heartbeat >= _HEARTBEAT_INTERVAL:
_last_heartbeat = _hb_now
_waiting_secs = int(_hb_now - last_chunk_time["t"])
if _waiting_secs >= _HEARTBEAT_INTERVAL:
# No chunks for 30s+ — rewrite the live spinner/status line
# so CLI/TUI/Desktop users see WHAT the wait is (slow or
# overloaded provider / long thinking pause) instead of an
# unexplained generic spinner, and WHEN recovery kicks in.
if (
_stream_stale_timeout is not None
and _stream_stale_timeout != float("inf")
):
_recovery = f"; auto-reconnect at {int(_stream_stale_timeout)}s"
else:
_recovery = ""
agent._emit_wait_notice(
f"⏳ waiting on {api_kwargs.get('model', 'the provider')} — "
f"{_waiting_secs}s with no output yet (provider may be "
f"slow or overloaded, or the model is thinking{_recovery})"
)
else:
_recovery = ""
# Chunks are flowing — keep the activity tracker fresh but
# leave the live display alone.
agent._touch_activity(
f"waiting for stream response ({_waiting_secs}s, no chunks yet)"
)
# Detect stale streams: connections kept alive by SSE pings
# but delivering no real chunks. Kill the client so the
# inner retry loop can start a fresh connection.
_stale_elapsed = time.time() - last_chunk_time["t"]
if _stale_elapsed > _stream_stale_timeout:
_est_ctx = estimate_request_context_tokens(api_kwargs)
logger.warning(
"Stream stale for %.0fs (threshold %.0fs) — no chunks received. "
"model=%s context=~%s tokens. Killing connection.",
_stale_elapsed, _stream_stale_timeout,
api_kwargs.get("model", "unknown"), f"{_est_ctx:,}",
)
agent._buffer_status(
f"⚠️ No response from provider for {int(_stale_elapsed)}s "
f"(model: {api_kwargs.get('model', 'unknown')}, "
f"context: ~{_est_ctx:,} tokens). "
f"Reconnecting..."
)
try:
_cancel_current_stream_attempt("stale_stream_kill")
_close_request_client_once("stale_stream_kill")
except Exception:
pass
# Circuit breaker (#58962): count the stale kill. See the
# canonical comment block above ``_stale_streak()``.
_bump_stale_streak(agent)
# Rebuild the primary client too — its connection pool
# may hold dead sockets from the same provider outage.
if agent.api_mode == "anthropic_messages":
# #67142: the stale stream ran on a request-local anthropic
# client, already socket-aborted above via
# _close_request_client_once (which unblocks the worker and
# preserves the #28161 no-hang guarantee). The shared
# _anthropic_client is NOT the in-flight transport, so we must
# not close it from this poll (stranger) thread — that was the
# FD-recycle corruption vector. Nothing further is needed.
pass
else:
# #70773: same FD-recycle corruption vector as #67142.
# The shared OpenAI client's connection pool must NOT be
# closed from this watchdog/poll thread — worker threads
# from previous stale-killed attempts may still be
# unwinding their SSL BIOs. The request-local client is
# already closed above via _close_request_client_once.
# The shared client will be replaced lazily by
# _ensure_primary_openai_client on the next request.
pass
# Reset the timer so we don't kill repeatedly while
# the inner thread processes the closure.
last_chunk_time["t"] = time.time()
agent._emit_wait_notice(
f"⏳ waiting on {api_kwargs.get('model', 'the provider')} — "
f"{_waiting_secs}s with no output yet (provider may be "
f"slow or overloaded, or the model is thinking{_recovery})"
f"⚠ no output from provider for {int(_stale_elapsed)}s — "
f"reconnecting..."
)
else:
# Chunks are flowing — keep the activity tracker fresh but
# leave the live display alone.
agent._touch_activity(
f"waiting for stream response ({_waiting_secs}s, no chunks yet)"
f"stale stream detected after {int(_stale_elapsed)}s, reconnecting"
)
# Detect stale streams: connections kept alive by SSE pings
# but delivering no real chunks. Kill the client so the
# inner retry loop can start a fresh connection.
_stale_elapsed = time.time() - last_chunk_time["t"]
if _stale_elapsed > _stream_stale_timeout:
_est_ctx = estimate_request_context_tokens(api_kwargs)
logger.warning(
"Stream stale for %.0fs (threshold %.0fs) — no chunks received. "
"model=%s context=~%s tokens. Killing connection.",
_stale_elapsed, _stream_stale_timeout,
api_kwargs.get("model", "unknown"), f"{_est_ctx:,}",
)
agent._buffer_status(
f"⚠️ No response from provider for {int(_stale_elapsed)}s "
f"(model: {api_kwargs.get('model', 'unknown')}, "
f"context: ~{_est_ctx:,} tokens). "
f"Reconnecting..."
)
try:
_cancel_current_stream_attempt("stale_stream_kill")
_close_request_client_once("stale_stream_kill")
except Exception:
pass
# Circuit breaker (#58962): count the stale kill. See the
# canonical comment block above ``_stale_streak()``.
_bump_stale_streak(agent)
# Rebuild the primary client too — its connection pool
# may hold dead sockets from the same provider outage.
if agent.api_mode == "anthropic_messages":
# #67142: the stale stream ran on a request-local anthropic
# client, already socket-aborted above via
# _close_request_client_once (which unblocks the worker and
# preserves the #28161 no-hang guarantee). The shared
# _anthropic_client is NOT the in-flight transport, so we must
# not close it from this poll (stranger) thread — that was the
# FD-recycle corruption vector. Nothing further is needed.
pass
else:
# #70773: same FD-recycle corruption vector as #67142.
# The shared OpenAI client's connection pool must NOT be
# closed from this watchdog/poll thread — worker threads
# from previous stale-killed attempts may still be
# unwinding their SSL BIOs. The request-local client is
# already closed above via _close_request_client_once.
# The shared client will be replaced lazily by
# _ensure_primary_openai_client on the next request.
pass
# Reset the timer so we don't kill repeatedly while
# the inner thread processes the closure.
last_chunk_time["t"] = time.time()
agent._emit_wait_notice(
f"⚠ no output from provider for {int(_stale_elapsed)}s — "
f"reconnecting..."
)
agent._touch_activity(
f"stale stream detected after {int(_stale_elapsed)}s, reconnecting"
)
if agent._interrupt_requested:
# The stale branch above already counted this iteration when its
# deadline won the race; do not double-count a simultaneous stop.
if _stale_elapsed <= _stream_stale_timeout:
_record_interrupted_provider_wait(
agent,
_stale_elapsed,
response_started=deltas_were_sent["yes"],
if agent._interrupt_requested:
# The stale branch above already counted this iteration when its
# deadline won the race; do not double-count a simultaneous stop.
if _stale_elapsed <= _stream_stale_timeout:
_record_interrupted_provider_wait(
agent,
_stale_elapsed,
response_started=deltas_were_sent["yes"],
)
# Mark THIS request cancelled before force-closing so the worker's
# exception handler recognizes the forced transport error as a
# cancel and exits without retrying or surfacing a network error.
# (#6600)
_request_cancelled["value"] = True
logger.debug(
"Force-closing streaming httpx client due to interrupt "
"(not a network error)."
)
# Mark THIS request cancelled before force-closing so the worker's
# exception handler recognizes the forced transport error as a
# cancel and exits without retrying or surfacing a network error.
# (#6600)
_request_cancelled["value"] = True
logger.debug(
"Force-closing streaming httpx client due to interrupt "
"(not a network error)."
)
try:
_cancel_current_stream_attempt("stream_interrupt_abort")
# #67142: kind-aware — anthropic aborts the request-local
# client's socket from this poll thread; the shared
# _anthropic_client is never closed here.
_close_request_client_once("stream_interrupt_abort")
except Exception:
pass
# Wait for the worker to unwind Relay-managed stream scopes
# (physical LLM + deferred logical) before surfacing
# InterruptedError. Raising immediately lets turn teardown
# (finish_logical_calls / end_turn / close_session) race a
# still-open physical scope and corrupt the LIFO stack —
# "scope handle is not at the top of the stack" → CLI EIO /
# redraw storm (#81521). No-op when Relay managed execution
# is not live.
_join_worker_for_relay_teardown(t, label="Streaming")
raise InterruptedError("Agent interrupted during streaming API call")
try:
_cancel_current_stream_attempt("stream_interrupt_abort")
# #67142: kind-aware — anthropic aborts the request-local
# client's socket from this poll thread; the shared
# _anthropic_client is never closed here.
_close_request_client_once("stream_interrupt_abort")
except Exception:
pass
# Wait for the worker to unwind Relay-managed stream scopes
# (physical LLM + deferred logical) before surfacing
# InterruptedError. Raising immediately lets turn teardown
# (finish_logical_calls / end_turn / close_session) race a
# still-open physical scope and corrupt the LIFO stack —
# "scope handle is not at the top of the stack" → CLI EIO /
# redraw storm (#81521). No-op when Relay managed execution
# is not live. (Inline mode has no worker: the request runs
# on the caller's thread and has already unwound by the time
# the InterruptedError below is raised.)
if t is not None:
_join_worker_for_relay_teardown(t, label="Streaming")
_monitor_interrupted["yes"] = True
return
if _inline:
# Request on THIS thread; heartbeat / stale / interrupt monitor on a
# side thread that only ever aborts sockets (never dispatches).
monitor = threading.Thread(
target=_context_thread_target(_monitor_loop),
name="stream-inline-monitor",
daemon=True,
)
monitor.start()
try:
_run_call()
finally:
monitor.join(timeout=2.0)
else:
_monitor_loop()
if _monitor_interrupted["yes"]:
raise InterruptedError("Agent interrupted during streaming API call")
# Worker thread exited before the main thread's poll loop could check
# the interrupt flag. If the worker returned early due to an interrupt
# (e.g. _call_anthropic() detected _interrupt_requested and returned
+33 -1
View File
@@ -310,6 +310,7 @@ def _record_codex_app_server_compaction(
# Native compaction rewrote the provider-side context; the usage anchor's
# transcript snapshot no longer matches what will be sent. Invalidate it.
agent._usage_anchor = None
agent._turn_base_usage_anchor = None
agent._last_compaction_in_place = False
try:
@@ -1100,7 +1101,9 @@ def _consume_codex_event_stream(
* ``on_first_delta()`` — one-shot, fires on the first text delta only.
* ``on_event(event)`` — fires for every event before any other processing.
Used for watchdog activity, debug logging, anything wire-shape-agnostic.
* ``interrupt_check()`` — returns True to break the loop early.
* ``interrupt_check()`` — returns True to break the loop early, or raises
``TimeoutError`` / ``InterruptedError`` for request-retirement control
flow that must not be converted into a partial final response.
"""
collected_output_items: List[Any] = []
# output_index of each collected_output_items entry, appended in lockstep
@@ -1605,18 +1608,39 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
max_stream_retries = 1
# Accumulate streamed text so callers / compat shims can read it.
agent._codex_streamed_text_parts: list = []
# Retirement token for THIS request, installed by
# ``interruptible_api_call`` before it hands off to the worker thread. When
# a watchdog (TTFB / stream-idle / stale-call) kills the connection it
# clears the agent-level token, so a worker that is still draining frames
# can tell it has been retired. ``None`` means no watchdog owns this call
# (auxiliary callers drive this function directly) — then every check
# passes and behavior is unchanged.
request_token = getattr(agent, "_active_codex_stream_request_token", None)
def _request_is_current() -> bool:
if request_token is None:
return True
return getattr(agent, "_active_codex_stream_request_token", None) is request_token
def _on_text_delta(text: str) -> None:
if not _request_is_current():
return
agent._codex_streamed_text_parts.append(text)
agent._fire_stream_delta(text)
def _on_reasoning_delta(text: str) -> None:
if not _request_is_current():
return
agent._fire_reasoning_delta(text)
def _on_commentary_message(text: str) -> None:
if not _request_is_current():
return
agent._fire_streamed_codex_commentary(text)
def _on_event(event: Any) -> None:
if not _request_is_current():
return
# TTFB watchdog and activity touch — runs once per SSE event.
agent._codex_stream_last_event_ts = time.time()
agent._touch_activity("receiving stream response")
@@ -1720,6 +1744,14 @@ def run_codex_stream(agent, api_kwargs: dict, client: Any = None, on_first_delta
raise
def _interrupt_or_superseded() -> bool:
# A retired request must NOT break out of the consume loop: breaking
# returns the partial `final` (status defaults to "completed"), which
# the caller persists as a finished assistant turn. Raise so the
# watchdog's own TimeoutError is what the retry path sees.
if not _request_is_current():
raise TimeoutError(
"Codex Responses stream request retired before terminal response"
)
return bool(agent._interrupt_requested)
try:
+3 -3
View File
@@ -253,7 +253,7 @@ CODING_AGENT_GUIDANCE = (
"paths for the same flaw and fix the class, not just the reported site.\n"
"- When fixing linter/type errors on a file, stop after about three "
"attempts on the same file and ask the user rather than looping.\n"
"- Track multi-step work with `todo`. Reference code as `path:line` instead "
"- Track multi-step work with `todo_list`. Reference code as `path:line` instead "
"of pasting whole files.\n"
"\n"
"Respect the user's repo: don't commit, push, or rewrite history unless "
@@ -547,9 +547,9 @@ class RuntimeMode:
trailing: list[str] = []
if self.profile.guidance:
brief = self.profile.guidance
if valid_tool_names is not None and "todo" not in valid_tool_names:
if valid_tool_names is not None and "todo_list" not in valid_tool_names:
brief = brief.replace(
"- Track multi-step work with `todo`. Reference code as "
"- Track multi-step work with `todo_list`. Reference code as "
"`path:line` instead of pasting whole files.",
"- Reference code as `path:line` instead of pasting "
"whole files.",
+12 -1
View File
@@ -135,9 +135,20 @@ def compute_session_context_breakdown(
# after the response) and far more accurate than the heuristic total.
from agent.model_metadata import anchored_context_tokens
# Prefer the turn-base anchor (first response of the current turn): on
# reasoning models, later same-turn responses inflate prompt_tokens with
# replayed thinking that evaporates at the turn boundary, so anchoring on
# the LAST response makes the meter sawtooth. Fall back to the last-
# response anchor, then to measured/estimated figures.
anchored_used = anchored_context_tokens(
messages or [], getattr(agent, "_usage_anchor", None)
messages or [],
getattr(agent, "_turn_base_usage_anchor", None),
charge_stale_thinking=False,
)
if anchored_used is None:
anchored_used = anchored_context_tokens(
messages or [], getattr(agent, "_usage_anchor", None)
)
measured_used = int(getattr(comp, "last_prompt_tokens", 0) or 0) if comp else 0
if anchored_used is not None:
context_used = anchored_used
+104 -19
View File
@@ -210,6 +210,20 @@ _TRUNCATED_SUMMARY_MARKER = "finish_reason=length"
def _is_summary_access_or_quota_error(exc: Exception) -> bool:
"""Return True for non-retryable summary auth, permission, or quota errors."""
# A credential read that failed closed because no profile secret scope
# was active (multiplexed gateway, worker thread without the caller's
# ContextVars) is a missing-credential failure of our own making: the
# summary model cannot be reached until the spawn site is fixed, and a
# placeholder summary would only destroy the middle window for nothing.
# Classify it with the credential class so compress() preserves the
# session unchanged (#100849 bundle: every hygiene pass truncated).
try:
from agent.secret_scope import UnscopedSecretError
except Exception: # pragma: no cover - import guard
UnscopedSecretError = () # type: ignore[assignment]
if UnscopedSecretError and isinstance(exc, UnscopedSecretError):
return True
classified = classify_api_error(exc)
if classified.reason is FailoverReason.rate_limit:
return False
@@ -2172,7 +2186,7 @@ def _summarize_tool_result_unguarded(tool_name: str, tool_args: str, tool_conten
target = args.get("target", "?")
return f"[memory] {action} on {target}"
if tool_name == "todo":
if tool_name == "todo_list":
return "[todo] updated task list"
if tool_name == "clarify":
@@ -2225,11 +2239,11 @@ def _summarize_tool_result_unguarded(tool_name: str, tool_args: str, tool_conten
if tool_name == "text_to_speech":
return f"[text_to_speech] generated audio ({content_len:,} chars)"
if tool_name == "cronjob":
if tool_name == "cronjob_manage":
action = args.get("action", "?")
return f"[cronjob] {action}"
if tool_name == "process":
if tool_name == "process_manage":
action = args.get("action", "?")
sid = args.get("session_id", "?")
return f"[process] {action} session={sid}"
@@ -2642,6 +2656,7 @@ class ContextCompressor(ContextEngine):
self.get_active_compression_failure_cooldown()
self._load_fallback_compression_streak()
self._load_ineffective_compression_count()
self._load_anti_thrash_recovery_deadline()
self._load_proactive_prune_rearm_tokens()
def on_session_start(self, session_id: str, **kwargs) -> None:
@@ -2807,6 +2822,45 @@ class ContextCompressor(ContextEngine):
except Exception as exc:
logger.debug("compression ineffective count persist failed (non-sqlite): %s", exc)
def _load_anti_thrash_recovery_deadline(self) -> None:
"""Restore the durable recovery deadline (wall-clock epoch, #100185).
Missing/absent storage leaves the in-memory clock disarmed, so the
next blocked evaluation arms a full fresh window (#54923).
"""
session_db = getattr(self, "_session_db", None)
session_id = getattr(self, "_session_id", "")
getter = getattr(session_db, "get_compression_recovery_deadline", None)
if not session_id or not callable(getter):
return
try:
stored = getter(session_id)
self._anti_thrash_recovery_deadline = max(
0.0,
float(stored) if isinstance(stored, (int, float, str)) else 0.0,
)
except (TypeError, ValueError, sqlite3.Error) as exc:
logger.debug("compression recovery deadline lookup failed: %s", exc)
except Exception as exc:
logger.debug("compression recovery deadline lookup failed (non-sqlite): %s", exc)
def _set_anti_thrash_recovery_deadline(self, deadline: float) -> None:
"""Set the recovery deadline, persisting on change only (0 = disarmed)."""
if deadline == self._anti_thrash_recovery_deadline:
return
self._anti_thrash_recovery_deadline = deadline
session_db = getattr(self, "_session_db", None)
session_id = getattr(self, "_session_id", "")
setter = getattr(session_db, "set_compression_recovery_deadline", None)
if not session_id or not callable(setter):
return
try:
setter(session_id, deadline)
except sqlite3.Error as exc:
logger.debug("compression recovery deadline persist failed: %s", exc)
except Exception as exc:
logger.debug("compression recovery deadline persist failed (non-sqlite): %s", exc)
def _record_ineffective_compression_verdict(self, count: int) -> None:
"""Set the anti-thrash strike counter, keeping the durable copy in sync.
@@ -3900,9 +3954,17 @@ class ContextCompressor(ContextEngine):
except Exception as exc:
logger.debug("compression ineffective-count refresh failed: %s", exc)
def _automatic_compression_blocked(self) -> bool:
"""Return whether automatic compaction is in cooldown or tripped."""
if not self._automatic_compression_blocked_locally():
def _automatic_compression_blocked(self, *, ignore_cooldown: bool = False) -> bool:
"""Return whether automatic compaction is in cooldown or tripped.
``ignore_cooldown=True`` evaluates only the breakers that are NOT the
summary-failure cooldown. Used by provider-proven overflow recovery
(#100661): the provider already rejected the request, so waiting out
the cooldown just wedges the session — every turn defers and the next
failure extends the ladder. The overflow path gets one real attempt;
the ineffective/structural breakers still apply.
"""
if not self._automatic_compression_blocked_locally(ignore_cooldown=ignore_cooldown):
return False
# Blocked on the in-memory snapshot. Durable guard rows may have
# been cleared by another agent since bind_session_state() — a
@@ -3912,9 +3974,9 @@ class ContextCompressor(ContextEngine):
# local block outlive the durable state that justified it. The
# unblocked hot path above never pays for the DB reads.
self._refresh_durable_guards()
return self._automatic_compression_blocked_locally()
return self._automatic_compression_blocked_locally(ignore_cooldown=ignore_cooldown)
def _automatic_compression_blocked_locally(self) -> bool:
def _automatic_compression_blocked_locally(self, *, ignore_cooldown: bool = False) -> bool:
"""Evaluate the automatic-compaction gate on in-memory state only."""
# Do not trigger compression while the summary LLM is in cooldown.
# On a 429/transient failure _generate_summary() sets a cooldown and
@@ -3926,7 +3988,7 @@ class ContextCompressor(ContextEngine):
# force=True, which clears this cooldown in compress() before running,
# so it still retries immediately.
_cooldown_remaining = self._summary_failure_cooldown_until - time.monotonic()
if _cooldown_remaining > 0:
if _cooldown_remaining > 0 and not ignore_cooldown:
if not self.quiet_mode:
logger.debug(
"Compression deferred — summary LLM in cooldown for %.0fs more",
@@ -3964,21 +4026,34 @@ class ContextCompressor(ContextEngine):
# the worst case in the truly-incompressible state is one compaction
# attempt per recovery window — bounded, not thrash.
#
# The clock is armed lazily on the first BLOCKED evaluation rather
# than persisted at trip time: a fresh process that loads a durable
# tripped counter (#69872) therefore starts a full window blocked,
# preserving the restart-must-not-disarm contract (#54923).
# The clock is armed lazily on the first BLOCKED evaluation and
# persisted on the session row (#100185): a fresh process/compressor
# that loads a durable tripped counter (#69872) with no stored
# deadline starts a full window blocked, preserving the
# restart-must-not-disarm contract (#54923) — but one that loads an
# already-armed deadline resumes that window instead of restarting it.
if (
self._ineffective_compression_count >= 2
or self._fallback_compression_streak >= 2
):
_now = time.monotonic()
if self._anti_thrash_recovery_deadline <= 0.0:
self._anti_thrash_recovery_deadline = (
# Wall clock, not monotonic: the deadline is persisted on the
# session row (#100185) so a fresh compressor bound to the same
# session — the gateway rebuilds the AIAgent on every cache
# eviction — resumes the SAME window instead of restarting it.
# Without that, a blocked messaging session never earned its
# probe and stayed blocked forever.
_now = time.time()
if self._anti_thrash_recovery_deadline <= 0.0 or (
# Clock jumped backwards past a full window: never wait
# longer than one window from now.
self._anti_thrash_recovery_deadline - _now
> self._ANTI_THRASH_RECOVERY_SECONDS
):
self._set_anti_thrash_recovery_deadline(
_now + self._ANTI_THRASH_RECOVERY_SECONDS
)
elif _now >= self._anti_thrash_recovery_deadline:
self._anti_thrash_recovery_deadline = 0.0
self._set_anti_thrash_recovery_deadline(0.0)
if self._ineffective_compression_count >= 2:
self._record_ineffective_compression_verdict(1)
if self._fallback_compression_streak >= 2:
@@ -4009,7 +4084,7 @@ class ContextCompressor(ContextEngine):
# Guard not tripped (counters were cleared by an effective compaction
# or a fitting real-usage reading) — disarm any pending recovery clock
# so a LATER trip starts its own full window.
self._anti_thrash_recovery_deadline = 0.0
self._set_anti_thrash_recovery_deadline(0.0)
return False
# ------------------------------------------------------------------
@@ -4946,6 +5021,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
turns_to_summarize: List[Dict[str, Any]],
focus_topic: Optional[str] = None,
memory_context: str = "",
bypass_cooldown: bool = False,
) -> Optional[str]:
"""Generate a structured summary of conversation turns.
@@ -4968,7 +5044,10 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb
if self._compression_cancelled():
raise AuxiliaryExplicitCancellation()
now = prompt_started_at
if now < self._summary_failure_cooldown_until:
# bypass_cooldown (#100661): provider-proven overflow gets ONE real
# summary attempt while the cooldown is armed; a failure below still
# records/extends the cooldown normally.
if now < self._summary_failure_cooldown_until and not bypass_cooldown:
logger.debug(
"Skipping context summary during cooldown (%.0fs remaining)",
self._summary_failure_cooldown_until - now,
@@ -7662,6 +7741,7 @@ This compaction should PRIORITISE preserving all information related to the focu
focus_topic: Optional[str] = None,
force: bool = False,
memory_context: str = "",
bypass_cooldown: bool = False,
) -> List[Dict[str, Any]]:
"""Compress conversation messages by summarizing middle turns.
@@ -7698,6 +7778,10 @@ This compaction should PRIORITISE preserving all information related to the focu
summary path. Auto-compress callers pass False.
memory_context: Optional provider-supplied context to preserve in
the summary prompt. Whitespace-only values are ignored.
bypass_cooldown: If True, run the summary LLM even while the
summary-failure cooldown is armed, WITHOUT clearing it
(#100661). Set by provider-proven overflow recovery, which
is already bounded by the caller's attempt budget.
"""
# Reset per-call summary failure state — callers inspect these fields
# after compress() returns to decide whether to surface a warning.
@@ -8035,6 +8119,7 @@ This compaction should PRIORITISE preserving all information related to the focu
turns_to_summarize,
focus_topic=summary_focus_topic,
memory_context=memory_context,
bypass_cooldown=bypass_cooldown,
)
except AuxiliaryExplicitCancellation:
# Explicit cancellation is a true no-op. Restore state mutated by
+203 -3
View File
@@ -718,6 +718,17 @@ class CompressionCommitFence:
self._progress_observed = False
self._deadline: float | None = None
self._retain_cancelled_lock_until_worker_done = False
# #97963: set by the worker (mark_commit_watermark_fenced) once its
# commit path is watermark-fenced — i.e. it captured the session's
# active-row watermark at compression start, so any row appended
# AFTER that point survives a late commit verbatim as concurrent
# tail (archive_and_compact / publish_compression_child clone rows
# above the watermark instead of archiving them). Hosts read this
# at the turn-hold boundary to decide whether a detached worker may
# KEEP its commit admission (safe: newer turns cannot be clobbered)
# or must be cancelled as before (unfenced commit; discard is the
# only safe outcome). Plain bool store — atomic in CPython.
self._commit_watermark_fenced = False
if total_ceiling_seconds is not None:
self.set_total_ceiling_seconds(total_ceiling_seconds)
@@ -748,6 +759,20 @@ class CompressionCommitFence:
deadline = self._deadline
return deadline is not None and time.monotonic() >= deadline
@property
def deadline_monotonic(self) -> float | None:
"""The armed deadline as an absolute ``time.monotonic()`` instant.
:meth:`set_total_ceiling_seconds` documents this deadline as "shared by
the host and worker", but until #99692 only the host could read it —
``deadline_exceeded`` answers "is it past?" for a caller that is already
polling, which is useless to a worker blocked inside a provider stream.
Publishing the instant itself lets the worker's stream consumer stop at
exactly the moment the host stops waiting (see
``auxiliary_client.aux_stream_deadline``).
"""
return self._deadline
def seconds_since_progress(self) -> float:
"""Seconds since the worker last reported forward progress."""
return max(0.0, time.monotonic() - self._last_progress)
@@ -843,6 +868,24 @@ class CompressionCommitFence:
"""Prevent a timed-out live worker from overlapping a retry."""
self._retain_cancelled_lock_until_worker_done = True
def mark_commit_watermark_fenced(self) -> None:
"""Record that this attempt's commit is bounded by a start watermark.
Called by the compression worker right after it captures
``get_active_message_watermark()`` under the durable compression
lock (#75316/#87484). A watermark-fenced commit archives ONLY rows
at or below the watermark; rows appended later — e.g. the user turn
the host released at the turn-hold boundary (#97963) — are cloned
as live concurrent tail. That is exactly the property a host needs
before letting a detached worker keep its commit admission.
"""
self._commit_watermark_fenced = True
@property
def commit_watermark_fenced(self) -> bool:
"""Lock-free read: the worker's commit is watermark-bounded."""
return self._commit_watermark_fenced
def allow_cancelled_lock_release(self) -> None:
"""Undo :meth:`retain_compression_lock_until_worker_done`.
@@ -1969,6 +2012,25 @@ def context_compression_timed_out(agent: Any) -> bool:
return getattr(agent, "_last_compression_timed_out", None) is True
def _automatic_gate_blocked(
blocked: Any, compressor: Any, bypass_cooldown: bool
) -> bool:
"""Evaluate the automatic breaker gate, optionally ignoring the cooldown.
Provider-proven overflow recovery (#100661) passes ``bypass_cooldown``;
engines whose gate predates the kwarg (plugins, test doubles) are called
with the legacy no-argument shape.
"""
if bypass_cooldown:
try:
accepts = "ignore_cooldown" in inspect.signature(blocked).parameters
except (TypeError, ValueError):
accepts = False
if accepts:
return bool(blocked(compressor, ignore_cooldown=True))
return bool(blocked(compressor))
def compression_blocked_transiently(agent: Any) -> bool:
"""Type-pinned read of the transient-block signal (#97488).
@@ -2205,6 +2267,7 @@ def _supported_compression_kwargs(
focus_topic: Optional[str],
force: bool,
memory_context: str,
bypass_cooldown: bool = False,
) -> dict:
"""Return only compression kwargs accepted by an engine callable.
@@ -2218,6 +2281,8 @@ def _supported_compression_kwargs(
"focus_topic": focus_topic,
"force": force,
}
if bypass_cooldown:
candidates["bypass_cooldown"] = True
if memory_context:
candidates["memory_context"] = memory_context
try:
@@ -2826,6 +2891,79 @@ def _is_real_user_message(message: Any) -> bool:
return not ContextCompressor._is_synthetic_compression_user_turn(message)
def _message_contains_busy_steer(message: Any) -> bool:
"""Return whether *message* carries a busy-steer marker.
With ``display.busy_input_mode: steer`` the follow-up is embedded as an
out-of-band marker inside a ``role=tool`` result (see
``agent_runtime_helpers.apply_pending_steer_to_tool_results``). That marker
carries real user intent but lives outside ``role=user``, so the
``_is_real_user_message`` / ``_transcript_has_real_user_turn`` checks
alone would miss it.
"""
text = _message_text(message)
if not text:
return False
try:
from agent.prompt_builder import STEER_MARKER_CLOSE, STEER_MARKER_OPEN
return STEER_MARKER_OPEN in text and STEER_MARKER_CLOSE in text
except Exception:
return "[OUT-OF-BAND USER MESSAGE" in text and "[/OUT-OF-BAND USER MESSAGE]" in text
def _extract_steer_text_from_message(message: Any) -> Optional[str]:
"""Extract the inner user text from a steer marker, or None."""
text = _message_text(message)
if not text:
return None
try:
from agent.prompt_builder import STEER_MARKER_CLOSE, STEER_MARKER_OPEN
open_marker = STEER_MARKER_OPEN
close_marker = STEER_MARKER_CLOSE
except Exception:
open_marker = "[OUT-OF-BAND USER MESSAGE"
close_marker = "[/OUT-OF-BAND USER MESSAGE]"
start = text.find(open_marker)
if start == -1:
# Fallback: marker wording may evolve; look for the stable prefix.
fallback_open = "[OUT-OF-BAND USER MESSAGE"
start = text.find(fallback_open)
if start == -1:
return None
# Skip to end of the opening line.
nl = text.find("\n", start)
if nl != -1:
start = nl + 1
else:
start += len(fallback_open)
else:
start += len(open_marker)
end = text.find(close_marker, start)
if end == -1:
end = text.find("[/OUT-OF-BAND USER MESSAGE]", start)
if end == -1:
return None
extracted = text[start:end].strip()
return extracted if extracted else None
def _compressed_has_busy_steer(messages: list) -> bool:
"""Whether *messages* already carries a steer marker (intent present).
Only ``role=tool`` rows count: that is the sole place the runtime ever
delivers a steer, so a compaction summary that merely quotes the marker
text must not be mistaken for live intent.
"""
for msg in messages:
if not isinstance(msg, dict) or msg.get("role") != "tool":
continue
if _message_contains_busy_steer(msg):
return True
return False
def _strip_stale_todo_snapshot(content: Any) -> Any:
"""Remove a previously merged todo-snapshot block from message content.
@@ -3028,17 +3166,34 @@ def _ensure_compressed_has_user_turn(
"""Preserve human intent, not merely a synthetic user-role placeholder."""
if any(_is_real_user_message(message) for message in compressed):
return "already_present"
if _compressed_has_busy_steer(compressed):
return "already_present"
from agent.context_compressor import (
COMPRESSION_CONTINUATION_USER_CONTENT,
_fresh_compaction_message_copy,
)
# One reversed positional scan: the anchor is whichever intent-bearing
# row is LAST in the original transcript — a real ``role=user`` turn or
# a steer marker riding inside a ``role=tool`` result. Scanning the two
# kinds separately (steer first, then user) would let an older, already
# consumed steer outrank a newer real user request and replay it
# (#100053 follow-up: ``[user A, tool(steer B), ..., user C]`` must
# anchor C, not B).
for message in reversed(original_messages):
if _is_real_user_message(message):
return _insert_real_user_anchor(
compressed,
_fresh_compaction_message_copy(message),
)
if not isinstance(message, dict) or message.get("role") != "tool":
continue
steer_text = _extract_steer_text_from_message(message)
if steer_text:
return _insert_real_user_anchor(
compressed,
{"role": "user", "content": steer_text},
)
from agent.message_metadata import append_message
append_message(
@@ -3156,6 +3311,7 @@ def compress_context(
task_id: str = "default",
focus_topic: Optional[str] = None,
force: bool = False,
bypass_cooldown: bool = False,
defer_context_engine_notification: bool = False,
commit_fence: Optional[CompressionCommitFence] = None,
) -> Tuple[list, str]:
@@ -3175,6 +3331,13 @@ def compress_context(
by the manual ``/compress`` slash command so users can retry
immediately after an auto-compress abort. Auto-compress
callers use the default ``False``.
bypass_cooldown: If True, the automatic breaker gates ignore ONLY the
summary-failure cooldown for this attempt (#100661). Set by the
provider-proven overflow recovery path: the provider already
rejected the request, so deferring until the cooldown lapses
wedges the session. Unlike ``force`` it does not clear the
cooldown, and the ineffective/structural breakers still apply;
a failed attempt records its cooldown normally.
defer_context_engine_notification: Delay the existing context-engine
hook until a manual host commits its outer history transaction.
commit_fence: Optional cooperative fence for executor callers that
@@ -3292,7 +3455,9 @@ def compress_context(
"_automatic_compression_blocked",
None,
)
if callable(blocked) and blocked(agent.context_compressor):
if callable(blocked) and _automatic_gate_blocked(
blocked, agent.context_compressor, bypass_cooldown
):
_mark_compression_blocked_transient(agent, agent.context_compressor)
existing_prompt = getattr(agent, "_cached_system_prompt", None)
if not existing_prompt:
@@ -3516,6 +3681,18 @@ def compress_context(
_commit_watermark = _lock_db.get_active_message_watermark(
_lock_sid
)
# #97963: a captured watermark makes the eventual
# commit safe against rows appended after this
# point (they survive as cloned concurrent tail on
# BOTH commit paths — archive_and_compact and
# publish_compression_child). Tell the fence so a
# host at the turn-hold boundary can keep this
# attempt's commit admission instead of burning it.
if commit_fence is not None:
try:
commit_fence.mark_commit_watermark_fenced()
except AttributeError:
pass # test doubles without the method
except Exception as _wm_err:
# Watermark capture is safety-additive: without it the
# commit falls back to archive-everything (historical
@@ -3751,7 +3928,9 @@ def compress_context(
"_automatic_compression_blocked",
None,
)
if callable(blocked) and blocked(compressor):
if callable(blocked) and _automatic_gate_blocked(
blocked, compressor, bypass_cooldown
):
_mark_compression_blocked_transient(agent, compressor)
_release_lock()
existing_prompt = getattr(agent, "_cached_system_prompt", None)
@@ -3948,6 +4127,7 @@ def compress_context(
focus_topic=focus_topic,
force=force,
memory_context=memory_context,
bypass_cooldown=bypass_cooldown,
)
if memory_context.strip() and "memory_context" not in compress_kwargs:
engine_name = getattr(
@@ -3992,11 +4172,28 @@ def compress_context(
from agent.auxiliary_client import (
aux_interrupt_protection,
aux_progress_hook,
aux_stream_deadline,
)
_progress_hook = (
commit_fence.touch_progress if commit_fence is not None
else (lambda: None)
)
# #99692: the progress hook above is the worker -> host leg; this is the
# return leg. _compression_cancel_requested (below) releases the compression
# OWNER when the host gives up, but the isolated provider daemon that
# actually holds the socket keeps streaming to its own budget —
# ``_aux_stream_total_ceiling`` = max(600, 4 * aux_timeout), which is >=
# the host's total ceiling for every configured timeout and starts
# counting later (after admission, serialization, prompt build and TTFT).
# With ``auxiliary.compression.timeout: 600`` that is 2400s of an
# orphaned 500K-token summary the commit fence is already guaranteed to
# refuse: paid tokens, a pinned HTTP connection, and — since every new
# turn re-triggers compression on a session that never shrank — a fresh
# orphan stacked on top of the last one. Sharing the host's absolute
# deadline makes the stream stop when the host it serves stops waiting.
_host_stream_deadline = (
commit_fence.deadline_monotonic if commit_fence is not None else None
)
# F4 state-ordering (#76354): a LATE successful summary must not undo
# the timeout cooldown the host recorded. Install a cancellation
# check the compressor consults BEFORE clearing the failure cooldown;
@@ -4042,7 +4239,9 @@ def compress_context(
)
compressed = messages
else:
with aux_progress_hook(_progress_hook), aux_interrupt_protection(
with aux_progress_hook(_progress_hook), aux_stream_deadline(
_host_stream_deadline
), aux_interrupt_protection(
cancel_check=_compression_cancel_requested
):
compressed = compress_fn(messages, **compress_kwargs)
@@ -5355,6 +5554,7 @@ def compress_context(
# the next response with usage re-anchors (its structural id/index
# check would also fail closed, but explicit is safer).
agent._usage_anchor = None
agent._turn_base_usage_anchor = None
# Arm the effectiveness verdict only after a completed rewrite crosses
# the full compaction boundary. Exceptions, aborts, and no-op attempts
# leave this false, so unrelated later usage cannot be charged to an
+296 -18
View File
@@ -41,6 +41,7 @@ from agent.conversation_compression import (
from agent.context_engine import automatic_compaction_status_message
from agent.display import KawaiiSpinner
from agent.error_classifier import FailoverReason, classify_api_error
from agent.fast_mode import begin_turn as begin_fast_mode_turn
from agent.message_metadata import append_message
from agent.turn_context import (
PreflightCompressionTimedOut,
@@ -652,6 +653,40 @@ def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str
)
def _maybe_grow_local_window(agent: Any, compressor: Any,
request_tokens: int) -> Optional[int]:
"""Try growing the managed local model's context window before
compressing. Returns the new window when the ladder granted one, else
None (hold / at native / not a managed local session).
The window ladder's design order: models launch at their zero-spill
window and grow toward native max as the session needs room;
compression is the move of last resort. Cheap for every non-local
provider: one lowercase compare, no imports.
"""
provider = (getattr(agent, "provider", "") or "").strip().lower()
if provider not in ("llamacpp", "llama.cpp", "llama-cpp", "custom"):
return None
base_url = getattr(agent, "base_url", "") or ""
if "127.0.0.1" not in base_url and "localhost" not in base_url:
return None
try:
from hermes_cli.local_runtime.growth import maybe_grow_window
current_window = int(getattr(compressor, "context_length", 0) or 0)
if current_window <= 0:
return None
return maybe_grow_window(
getattr(agent, "model", "") or "",
base_url=base_url,
session_tokens=int(request_tokens),
current_window=current_window,
)
except Exception as exc: # noqa: BLE001 — growth must never break a turn
logger.debug("local window growth check failed: %s", exc)
return None
def _ra():
"""Lazy reference to ``run_agent`` so callers can patch
``run_agent.handle_function_call`` / ``run_agent._set_interrupt`` /
@@ -954,6 +989,7 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history)
"""
stored_prompt = None
stored_state = "missing"
session_row = None
if conversation_history and agent._session_db:
try:
session_row = agent._session_db.get_session(agent.session_id)
@@ -1056,6 +1092,17 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history)
# Continuing session — reuse the exact system prompt from the
# previous turn so the Anthropic cache prefix matches.
agent._cached_system_prompt = stored_prompt
# Same contract for tools[]: a fresh AIAgent for an existing session
# (gateway agent-cache eviction) re-probed every check_fn, so pin the
# array back to the order this session already sent (tools freeze).
try:
saved_tools = session_row.get("tool_names") if session_row else None
if saved_tools:
from tools.mcp_tool import restore_agent_tool_prefix
restore_agent_tool_prefix(agent, json.loads(saved_tools))
except Exception:
logger.debug("tool prefix restore skipped", exc_info=True)
# Prompt-section callbacks are new-session-only. Recover their frozen
# bytes from the persisted full prompt so a later compression rebuild
# keeps them without evaluating plugin state in this resumed process.
@@ -1138,6 +1185,9 @@ def _restore_or_build_system_prompt(agent, system_message, conversation_history)
if agent._session_db:
try:
agent._session_db.update_system_prompt(agent.session_id, agent._cached_system_prompt)
from tools.mcp_tool import persist_agent_tool_names
persist_agent_tool_names(agent)
except Exception as exc:
logger.warning(
"Session DB update_system_prompt failed for session %s: "
@@ -1590,6 +1640,41 @@ def _compression_deferred_result(
}
def _provider_overflow_exhausted_result(
agent,
messages: List[Dict],
conversation_history,
api_call_count: int,
request_pressure_tokens: int,
max_compression_attempts: int,
) -> Dict[str, Any]:
"""Fail closed when a rebuilt request is still too large after recovery."""
agent._flush_status_buffer()
logger.error(
"%sContext compression failed after %d attempts; rebuilt request "
"remains over threshold at ~%s tokens.",
agent.log_prefix,
max_compression_attempts,
f"{request_pressure_tokens:,}",
)
agent._persist_session(messages, conversation_history)
final_response = (
"Context length exceeded: compression could not reduce the rebuilt "
"request below the safe threshold."
)
return {
"final_response": final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
"turn_exit_reason": "context_compression_exhausted",
}
def _rewrite_system_content_blocks(system_message: dict, effective: str) -> bool:
"""Rewrite a cache-decorated system message in place, keeping its blocks.
@@ -1973,6 +2058,7 @@ def run_conversation(
agent._last_compaction_in_place = False
agent._last_compression_attempt_recorded = False
agent._last_compression_attempt_in_place = None
begin_fast_mode_turn(agent, conversation_history)
# Adopt any ~/.hermes/.env credential/base-url edits made since the last
# turn — a Settings save updates .env but not this worker's client, which
@@ -2091,6 +2177,12 @@ def run_conversation(
failed = False
codex_ack_continuations = 0
length_continue_retries = 0
# One-shot "continue without thinking" override is turn-scoped: a
# thinking-only truncation arms it right before the continuation restart,
# and build_api_kwargs consumes it on that call. If the turn is
# interrupted/errors between arm and consume, it must not fire on the
# next turn's first request.
agent._ephemeral_reasoning_off = False
# Total outer-loop exceptions this turn (#92450) — see _MAX_OUTER_LOOP_ERRORS.
_outer_error_count = 0
truncated_tool_call_retries = 0
@@ -2108,6 +2200,13 @@ def run_conversation(
max_compression_attempts = getattr(agent, "max_compression_attempts", 3)
_last_preflight_pressure: Optional[int] = None
_preflight_compression_blocked = _ctx.preflight_compression_blocked
# A provider overflow is stronger evidence than the rough-estimate
# calibration that normally defers preflight immediately after compaction.
# Keep recovery armed until the rebuilt, complete request is below the
# configured compression threshold. Without this handoff, a compaction
# that drops rows but grows the actual prompt can be sent straight back to
# the provider while awaiting_real_usage_after_compression is true.
_provider_overflow_recovery_pending = False
# Armed when a compression host-timeout terminates the turn (#98722,
# salvaged from #98741); finalize below reuses the gateway's existing
# context-recovery contract (error/partial/compression_exhausted).
@@ -2832,6 +2931,21 @@ def run_conversation(
_preflight_threshold = int(
getattr(_compressor, "threshold_tokens", 0) or 0
)
_provider_overflow_preflight = (
_provider_overflow_recovery_pending
and (
_preflight_threshold <= 0
or request_pressure_tokens >= _preflight_threshold
)
)
if (
_provider_overflow_recovery_pending
and not _provider_overflow_preflight
):
# The outer-loop rebuild includes the active system prompt,
# request-only injections, and tool schemas. Once that complete
# request has real output runway again, the provider may be tried.
_provider_overflow_recovery_pending = False
# A previous mid-turn preflight pass deliberately continued the loop so
# API-only context and all sanitization could be rebuilt. Compare that
# fully assembled request with the fully assembled request that caused
@@ -2871,11 +2985,50 @@ def run_conversation(
and not _review_fork_first_request_pending(agent)
and len(messages) > 1
and compression_attempts < max_compression_attempts
and not _preflight_compression_blocked
and not _defer_preflight(request_pressure_tokens)
and (
not _preflight_compression_blocked
or _provider_overflow_preflight
)
and (
not _defer_preflight(request_pressure_tokens)
or _provider_overflow_preflight
)
and not _compression_cooldown
and _compressor.should_compress(request_pressure_tokens)
):
# Managed local runtime: try GROWING the context window before
# compressing (the window ladder's design order — compression is
# the move of last resort, once the window is at the model's
# native max or physics/speed say stop). Only fires for a
# llamacpp-flavored provider whose base_url is the server this
# process supervises; every other provider falls straight
# through to compression, exactly as before.
_grown_window = _maybe_grow_local_window(
agent, _compressor, request_pressure_tokens
)
if _grown_window:
# The server now grants a bigger window: recalibrate the
# compressor to it and skip compression this pass — the
# request that was over the OLD threshold fits the new one.
_compressor.update_model(
agent.model,
_grown_window,
base_url=getattr(agent, "base_url", "") or "",
api_key=getattr(agent, "api_key", "") or "",
provider=getattr(agent, "provider", "") or "",
api_mode=getattr(agent, "api_mode", "") or "",
)
agent._buffer_status(
f"📈 Context window grown to {_grown_window // 1024}K "
f"(local model; conversation continues uncompressed)"
)
# This preflight iteration never reached the provider —
# refund the consumed call/budget exactly as the compression
# path below does before ITS continue.
api_call_count -= 1
agent._api_call_count = api_call_count
agent.iteration_budget.refund()
continue
if _moa_prepared_request is not None:
pending_moa_prepared_request = _moa_prepared_request
compression_attempts += 1
@@ -3006,6 +3159,34 @@ def run_conversation(
_turn_exit_reason = "compaction_handoff_not_actionable"
break
continue
elif _provider_overflow_preflight and _compression_cooldown:
# The provider already proved this request cannot fit, while the
# compressor is temporarily unavailable. Do not send the known-
# oversized request again; let the next user turn retry after the
# cooldown instead of turning this into compression exhaustion.
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent,
messages,
api_call_count,
reason="transient_block",
)
elif (
_provider_overflow_preflight
and compression_attempts >= max_compression_attempts
):
# Every bounded recovery pass has been consumed and the rebuilt
# request is still over threshold. Fail closed before another
# provider call; llama.cpp can silently truncate an oversized
# retry instead of returning a second actionable overflow error.
return _provider_overflow_exhausted_result(
agent,
messages,
conversation_history,
api_call_count,
request_pressure_tokens,
max_compression_attempts,
)
elif (
agent.compression_enabled
and len(messages) > 1
@@ -3060,6 +3241,20 @@ def run_conversation(
if callable(_warn_fn):
_warn_fn(request_pressure_tokens, _ctx_len)
if _provider_overflow_preflight:
# Any other gate that prevented the forced preflight (for example,
# an uncompressible one-message request) must also fail closed.
# Falling through would send a request that the provider already
# proved cannot fit.
return _provider_overflow_exhausted_result(
agent,
messages,
conversation_history,
api_call_count,
request_pressure_tokens,
max_compression_attempts,
)
# Thinking spinner for quiet mode (animated during API call)
thinking_spinner = None
@@ -3991,7 +4186,7 @@ def run_conversation(
"The model used all its output tokens on reasoning "
"and had none left for the actual response.\n\n"
"To fix this:\n"
"→ Lower reasoning effort: `/thinkon low` or `/thinkon minimal`\n"
"→ Lower reasoning effort: `/reasoning low` or `/reasoning minimal`\n"
"→ Or switch to a larger/non-reasoning model with `/model`"
)
agent._cleanup_task_resources(effective_task_id)
@@ -4118,29 +4313,46 @@ def run_conversation(
)
if assistant_message is not None and not _trunc_has_tool_calls:
length_continue_retries += 1
# An EMPTY partial-stream stub (stream dropped
# mid tool-call before any text was delivered)
# must not be appended as an interim assistant
# message: it would serialize as
# {"role": "assistant", "content": ""}, and
# An interim assistant message with NO visible
# content must not be appended — whichever way it
# got that way. An empty partial-stream stub
# (stream dropped before any text was delivered)
# and a response whose whole output budget went to
# reasoning delivered in a separate field (GLM-5.3
# on ollama-cloud with reasoning_effort=high:
# finish_reason="length", content="",
# completion_tokens == max_tokens) both serialize
# as {"role": "assistant", "content": ""}, and
# strict providers (Moonshot/Kimi via OpenRouter)
# reject empty assistant content with HTTP 400
# ("message ... with role 'assistant' must not be
# empty") on the very next replay — permanently
# poisoning the session history. There is no
# partial text to continue from anyway, so only
# the continuation user-message is appended.
# poisoning the session history until the pre-call
# sanitizer "heals" the hole (observed 3+ healings
# per turn). There is no partial text to continue
# from anyway, so only the continuation
# user-message is appended.
_interim_content = getattr(assistant_message, "content", None)
_is_empty_partial_stub = (
getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID
and not getattr(assistant_message, "content", None)
and not _interim_content
)
if not _is_empty_partial_stub:
if not _interim_content and not _is_empty_partial_stub:
# Thinking-only truncation: the model spent the
# entire output cap on reasoning and produced no
# visible text. A continuation with thinking
# ON would re-think the whole context from
# scratch (continuations never replay prior
# reasoning) and re-burn the same budget, so
# the next call drops thinking for one request
# — the answer must be written, not re-derived.
agent._ephemeral_reasoning_off = True
if _interim_content:
interim_msg = agent._build_assistant_message(assistant_message, finish_reason)
# Marked so the ceiling exit can drop the fragment trail.
interim_msg["_length_continuation_fragment"] = True
append_message(messages, interim_msg)
if assistant_message.content:
truncated_response_parts.append(assistant_message.content)
truncated_response_parts.append(_interim_content)
if length_continue_retries < 4:
_is_partial_stream_stub = (
@@ -4184,13 +4396,43 @@ def run_conversation(
break
partial_response = agent._strip_think_blocks(_join_truncated_parts(truncated_response_parts)).strip()
# The pending one-shot reasoning-off override must
# not leak into the next turn when the 4th
# truncation goes straight to the ceiling exit
# without scheduling a continuation call to
# consume it.
agent._ephemeral_reasoning_off = False
if partial_response:
agent._vprint(
f"{agent.log_prefix}⚠️ Response still truncated "
f"after 4 continuation attempts — keeping the "
f"after {length_continue_retries} continuation attempts — keeping the "
f"partial response received so far.",
force=True,
)
_ceiling_final = partial_response
else:
# Every fragment was empty — e.g. a thinking
# model that spent each attempt's whole cap on
# reasoning (GLM-5.3 on ollama-cloud). Return
# an actionable message instead of an invisible
# None result, which only surfaces as a bare
# error card.
agent._vprint(
f"{agent.log_prefix}⚠️ Response still truncated "
f"after {length_continue_retries} continuation attempts — no visible "
f"text was produced.",
force=True,
)
_ceiling_final = (
"⚠️ **No visible answer was produced.** The "
"model hit its output-token limit on every "
"continuation attempt — its reasoning "
"consumed the entire budget each time.\n\n"
"To fix this:\n"
"→ Lower reasoning effort: `/reasoning low` "
"or `/reasoning none`\n"
"→ Or raise max_tokens for this model"
)
# Unanswered continue nudges made every later turn re-truncate.
_turn_start = (
current_turn_user_idx + 1
@@ -4218,7 +4460,7 @@ def run_conversation(
agent._cleanup_task_resources(effective_task_id)
agent._persist_session(messages, conversation_history)
return {
"final_response": partial_response or None,
"final_response": _ceiling_final,
"messages": messages,
"api_calls": api_call_count,
"completed": False,
@@ -4419,6 +4661,20 @@ def run_conversation(
)
if _new_anchor is not None:
agent._usage_anchor = _new_anchor
# Turn-base anchor for display surfaces: the FIRST
# response of a turn carries minimal current-turn
# reasoning replay, so its prompt_tokens approximate
# the durable transcript cost (what the next turn
# inherits). Later same-turn responses inflate
# prompt_tokens with replayed thinking + tool
# scaffolding that evaporates at the turn boundary —
# anchoring the context meter here instead of on the
# last response removes the end-of-turn sawtooth
# (850K mid-loop -> 600K next turn) that users read
# as a broken compaction. Display-only: compression
# trigger math keeps using real last-request usage.
if api_call_count == 1:
agent._turn_base_usage_anchor = _new_anchor
_compression_threshold = int(
getattr(agent.context_compressor, "threshold_tokens", 0)
or 0
@@ -5639,6 +5895,12 @@ def run_conversation(
)
)
time.sleep(2)
# Same class as the generic overflow handler below:
# the provider proved the request does not fit the
# (now-reduced) window, and row count alone is not
# proof the rebuilt request does. Recheck the
# complete request before the next provider call.
_provider_overflow_recovery_pending = True
_retry.restart_with_compressed_messages = True
break
# Fall through to normal error handling if compression
@@ -5924,6 +6186,11 @@ def run_conversation(
messages, system_message,
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
task_id=effective_task_id,
# #100661: the provider proved the request does not fit.
# Ignore the summary-failure cooldown for this ONE
# attempt (bounded by max_compression_attempts) instead
# of deferring every turn until the ladder lapses.
bypass_cooldown=True,
)
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
# #69870 lock-skip: the provider proved the request
@@ -6100,6 +6367,7 @@ def run_conversation(
messages, system_message,
approx_tokens=request_input_estimate,
task_id=effective_task_id,
bypass_cooldown=True, # #100661 provider-proven overflow
)
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
compression_attempts -= 1
@@ -6263,6 +6531,11 @@ def run_conversation(
messages, system_message,
approx_tokens=estimate_request_tokens_rough(api_messages, tools=agent.tools or None),
task_id=effective_task_id,
# #100661: the provider proved the request does not fit.
# Ignore the summary-failure cooldown for this ONE
# attempt (bounded by max_compression_attempts) instead
# of deferring every turn until the ladder lapses.
bypass_cooldown=True,
)
if messages is _overflow_input and compression_skipped_due_to_lock(agent):
# #69870 lock-skip: the provider proved the request
@@ -6326,6 +6599,11 @@ def run_conversation(
elif new_tokens > 0 and new_tokens < original_tokens * 0.95:
agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens))
time.sleep(2) # Brief pause between compression retries
# Rebuild the complete request before the next provider
# call and force normal preflight to honor it. Message
# count alone is not proof that system/tool-inclusive
# token pressure fell.
_provider_overflow_recovery_pending = True
_retry.restart_with_compressed_messages = True
break
else:
@@ -7632,7 +7910,7 @@ def run_conversation(
# This classification is needed regardless of whether the turn has visible content,
# because a substantive tool-only turn must invalidate any older housekeeping fallback.
_HOUSEKEEPING_TOOLS = frozenset({
"memory", "todo", "skill_manage", "session_search",
"memory", "todo_list", "skill_manage", "session_search",
})
_all_housekeeping = all(
tc.function.name in _HOUSEKEEPING_TOOLS
+112 -3
View File
@@ -9,6 +9,7 @@ back into the minimal shape Hermes expects from an OpenAI client.
from __future__ import annotations
import json
import logging
import os
import queue
import re
@@ -31,6 +32,7 @@ from agent.redact import redact_sensitive_text
from tools.environments.local import hermes_subprocess_env
ACP_MARKER_BASE_URL = "acp://copilot"
logger = logging.getLogger(__name__)
_DEFAULT_TIMEOUT_SECONDS = 900.0
# Stderr fingerprint of the deprecated `gh copilot` CLI extension
@@ -185,6 +187,71 @@ def _permission_denied(message_id: Any) -> dict[str, Any]:
}
def _model_selection_request(
session: dict[str, Any], requested_model: str
) -> tuple[str, dict[str, str]] | None:
"""Return the ACP request that selects ``requested_model`` for ``session``.
Prefer stable v1 ``session/set_config_option``. Fall back to Copilot's
pre-stabilization ``session/set_model`` extension only when no model
config option is advertised. A reported model list is authoritative:
unknown and policy-disabled ids return None instead of being sent.
"""
session_id = str(session.get("sessionId") or "").strip()
requested_model = str(requested_model or "").strip()
if not session_id or not requested_model or requested_model == "copilot-acp":
return None
config_options = [
o for o in (session.get("configOptions") or []) if isinstance(o, dict)
]
model_option = next(
(
o for o in config_options
if o.get("category") == "model" or o.get("id") == "model"
),
None,
)
if model_option is not None:
enabled_values = {
str(o.get("value") or "").strip()
for o in (model_option.get("options") or [])
if isinstance(o, dict)
and str(
((o.get("_meta") or {}).get("copilotEnablement")) or ""
).strip().lower() != "disabled"
}
if requested_model not in enabled_values:
return None
return (
"session/set_config_option",
{
"sessionId": session_id,
"configId": str(model_option.get("id") or "model"),
"value": requested_model,
},
)
advertised = [
m
for m in ((session.get("models") or {}).get("availableModels") or [])
if isinstance(m, dict)
]
available = {
str(m.get("modelId") or "").strip()
for m in advertised
if str(
((m.get("_meta") or {}).get("copilotEnablement")) or ""
).strip().lower() != "disabled"
}
if available and requested_model not in available:
return None
return (
"session/set_model",
{"sessionId": session_id, "modelId": requested_model},
)
def _format_messages_as_prompt(
messages: list[dict[str, Any]],
model: str | None = None,
@@ -197,8 +264,11 @@ def _format_messages_as_prompt(
"IMPORTANT: If you take an action with a tool, you MUST output tool calls using <tool_call>{...}</tool_call> blocks with JSON exactly in OpenAI function-call shape.",
"If no tool is needed, answer normally.",
]
if model:
sections.append(f"Hermes requested model hint: {model}")
# Deliberately no "requested model" line in the prompt: the model is
# applied for real via ACP session/set_model, and when the backend can't
# honor it (org-policy-disabled id) a prompt-text mention makes the
# serving model FALSELY self-identify as the requested one. Identity
# must come from the backend, not from prompt suggestion.
# Copilot has no tools of its own that would collide with Hermes', so it
# forwards the whole toolset (no allowlist).
@@ -365,6 +435,7 @@ class CopilotACPClient:
response_text, reasoning_text = self._run_prompt(
prompt_text,
timeout_seconds=_effective_timeout,
model=model,
)
tool_calls, cleaned_text = _extract_tool_calls_from_text(response_text)
@@ -393,7 +464,13 @@ class CopilotACPClient:
return _completion_to_stream_chunks(completion)
return completion
def _run_prompt(self, prompt_text: str, *, timeout_seconds: float) -> tuple[str, str]:
def _run_prompt(
self,
prompt_text: str,
*,
timeout_seconds: float,
model: str | None = None,
) -> tuple[str, str]:
# Fast-fail when the CLI doesn't support the ACP args we'd pass.
# Without this guard, a CLI like Claude Code v2.x exits with
# ``error: unknown option '--acp'`` immediately, then the parent
@@ -416,6 +493,13 @@ class CopilotACPClient:
f"to a working pair."
)
# Note the model Hermes selected; it is applied after session/new via
# the ACP-native `session/set_model` call. The CLI's `--model` spawn
# flag is deliberately NOT used here: `copilot --acp` validates it
# (an unknown id aborts the spawn) but then ignores it for the actual
# session, so it adds a failure mode without selecting anything.
requested_model = str(model or "").strip()
try:
# Hide the console the CLI child would otherwise flash on Windows
# (#56747). Hide-only — stdio pipes stay intact for the ACP wire.
@@ -560,6 +644,31 @@ class CopilotACPClient:
if not session_id:
raise RuntimeError("Copilot ACP did not return a sessionId.")
# Select the model Hermes asked for. Prefer the stable ACP v1
# session-config API: session/new advertises a category="model"
# select option and session/set_config_option updates it. Copilot
# still exposes the older models/session/set_model extension too,
# so retain that only as compatibility fallback for older agents.
if requested_model and requested_model != "copilot-acp":
try:
selection = _model_selection_request(session, requested_model)
if selection is not None:
method, params = selection
_request(method, params)
else:
logger.warning(
"Copilot ACP does not offer model %r; using the "
"session default.",
requested_model,
)
except Exception as exc:
logger.warning(
"Copilot ACP model selection for %r failed; continuing "
"with the session default: %s",
requested_model,
exc,
)
text_parts: list[str] = []
reasoning_parts: list[str] = []
_request(
+182 -10
View File
@@ -12,7 +12,7 @@ import re
from dataclasses import dataclass, fields, replace
from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional, Set, Tuple
from typing import Any, Dict, Iterable, List, Optional, Set, Tuple
from hermes_constants import OPENROUTER_BASE_URL
from hermes_cli.config import load_env
@@ -26,6 +26,7 @@ import hermes_cli.auth as auth_mod
from hermes_cli.auth import (
CODEX_ACCESS_TOKEN_REFRESH_SKEW_SECONDS,
PROVIDER_REGISTRY,
SINGLE_USE_REFRESH_POOL_PROVIDERS,
_auth_store_lock,
_codex_access_token_is_expiring,
_decode_jwt_claims,
@@ -807,11 +808,134 @@ def _write_through_provider_state_to_global_root(
)
def _singleton_target_for_entry(pool: "CredentialPool", entry: "PooledCredential") -> Optional[Path]:
"""Root ``.anthropic_oauth.json`` when *entry* is a borrowed hermes_pkce row, else None."""
if entry.source != "hermes_pkce" or entry.id not in getattr(pool, "_borrowed_root_ids", ()):
return None
try:
from agent.anthropic_credentials import _root_hermes_oauth_file
return _root_hermes_oauth_file()
except Exception:
return None
def _profile_owns_pool_provider(provider: str) -> bool:
"""True when the ACTIVE auth.json has its own rows for *provider*.
Named profiles with no local rows read the provider through the
``read_credential_pool`` global-root fallback ("borrowing").
"""
try:
pool = _load_auth_store().get("credential_pool")
except Exception:
return True # unreadable store: assume ownership, keep legacy path
entries = pool.get(provider) if isinstance(pool, dict) else None
return isinstance(entries, list) and bool(entries)
def _borrowed_single_use_pool_root() -> Optional[Path]:
"""Return the global-root auth.json when persisting a BORROWED single-use pool.
``None`` means "persist to the active store as usual": classic mode
(profile == root), or the profile owns its own rows for this provider.
Pytest seat belt mirrors ``_write_through_provider_state_to_global_root``.
"""
try:
global_path = _global_auth_file_path()
except Exception:
return None
if global_path is None:
return None
if os.environ.get("PYTEST_CURRENT_TEST"):
real_home_env = os.environ.get("HOME", "")
if real_home_env:
real_root = Path(real_home_env) / ".hermes" / "auth.json"
try:
if global_path.resolve(strict=False) == real_root.resolve(strict=False):
return None
except Exception:
return None
return global_path
def persist_pool_entries(
provider: str,
payloads: List[Dict[str, Any]],
*,
removed_ids: Optional[Iterable[str]] = None,
) -> None:
"""Persist a provider's pool rows to the store that OWNS them.
A named profile that sees a single-use-refresh provider (Anthropic,
Codex, xAI OAuth) only through the global-root fallback must not
materialize a local ``credential_pool.<provider>`` copy on its first
persist: that copy forks the single-use refresh token, the first profile
to rotate commits the new pair only to its own file, and root plus every
sibling die with ``invalid_grant`` on their next refresh (#100339). Such
rows are written back to the root store (under the root lock) so the
rotation is visible to every profile; everything else goes to the active
store exactly as before.
"""
if provider in SINGLE_USE_REFRESH_POOL_PROVIDERS and not _profile_owns_pool_provider(provider):
global_path = _borrowed_single_use_pool_root()
if global_path is not None:
removed = {rid for rid in (removed_ids or ()) if rid}
try:
with _auth_store_lock(target_path=global_path):
store = _load_auth_store(global_path)
pool = store.get("credential_pool")
if not isinstance(pool, dict):
pool = {}
store["credential_pool"] = pool
existing = pool.get(provider)
existing_list = existing if isinstance(existing, list) else []
incoming_by_id = {
p.get("id"): p for p in payloads
if isinstance(p, dict) and p.get("id")
}
# UPDATE-ONLY: a borrower may refresh the root's rows
# (rotation, cooldown state) but never add or delete
# them — the root owns their lifecycle. In particular a
# profile's singleton-prune (it has no
# .anthropic_oauth.json of its own) must not delete the
# root grant, and ``removed_ids`` is ignored here.
merged: List[Dict[str, Any]] = []
changed = False
for disk_entry in existing_list:
did = disk_entry.get("id") if isinstance(disk_entry, dict) else None
incoming = incoming_by_id.get(did) if did else None
if incoming is None:
merged.append(disk_entry)
continue
updated = auth_mod._merge_disk_cooldown_state(incoming, disk_entry, provider)
if updated != disk_entry:
changed = True
merged.append(updated)
if changed:
pool[provider] = merged
_save_auth_store(store, target_path=global_path)
return
except Exception as exc:
# Fail closed on the FORK, not on the save: never fall back to
# writing a local copy (that IS the bug). The in-memory pool
# still holds the rotated pair for this process.
logger.warning(
"%s pool: write-through of borrowed root grant failed (%s); "
"not materializing a profile-local copy",
provider, exc,
)
return
write_credential_pool(provider, payloads, removed_ids=removed_ids)
class CredentialPool:
def __init__(self, provider: str, entries: List[PooledCredential]):
self.provider = provider
self._entries = sorted(entries, key=lambda entry: entry.priority)
self._current_id: Optional[str] = None
# Ids of rows read via the global-root fallback (single-use OAuth
# providers only); set by load_pool(), consumed by add_entry().
self._borrowed_root_ids: Set[str] = set()
self._strategy = get_pool_strategy(provider)
# RLock: the mutation primitives below (_replace_entry/_persist)
# self-acquire this lock so the DEFERRED single-use-token refresh
@@ -938,7 +1062,7 @@ class CredentialPool:
# Self-locking (RLock): snapshotting self._entries must not race a
# concurrent rotation when called from the deferred refresh path.
with self._lock:
write_credential_pool(
persist_pool_entries(
self.provider,
[entry.to_dict() for entry in self._entries],
removed_ids=removed_ids,
@@ -1786,10 +1910,14 @@ class CredentialPool:
elif entry.source == "hermes_pkce":
try:
from agent.anthropic_credentials import _write_hermes_oauth_credentials
# A borrowed row was seeded from the ROOT's singleton
# (this profile has none); commit the rotation there,
# never into a new profile-local copy (#100339).
_write_hermes_oauth_credentials(
refreshed["access_token"],
refreshed["refresh_token"],
refreshed["expires_at_ms"],
target=_singleton_target_for_entry(self, entry),
)
except Exception as wexc:
# Same transaction rule as claude_code above.
@@ -2806,7 +2934,7 @@ class CredentialPool:
replace(entry, priority=new_priority)
for new_priority, entry in enumerate(self._entries)
]
write_credential_pool(
persist_pool_entries(
self.provider,
[entry.to_dict() for entry in self._entries],
removed_ids=[removed.id],
@@ -2845,7 +2973,22 @@ class CredentialPool:
with self._lock:
entry = replace(entry, priority=_next_priority(self._entries))
self._entries.append(entry)
self._persist()
borrowed_ids = getattr(self, "_borrowed_root_ids", None)
if borrowed_ids:
# ``hermes -p <profile> auth add <single-use provider>``: the
# profile is claiming its OWN credential. Persist only the
# profile-owned rows locally — copying the borrowed root
# grant alongside them would fork its single-use refresh
# token (#100339). Once the profile owns rows, the root
# fallback for this provider is shadowed (existing contract).
write_credential_pool(
self.provider,
[e.to_dict() for e in self._entries if e.id not in borrowed_ids],
)
self._entries = [e for e in self._entries if e.id not in borrowed_ids]
self._borrowed_root_ids = set()
else:
self._persist()
return entry
@@ -3631,6 +3774,12 @@ def _seed_custom_pool(pool_key: str, entries: List[PooledCredential]) -> Tuple[b
def load_pool(provider: str) -> CredentialPool:
provider = (provider or "").strip().lower()
if provider in SINGLE_USE_REFRESH_POOL_PROVIDERS:
# One-time heal for installs that forked this grant across profiles
# BEFORE the clone-strip / root-write-through existed: consolidate the
# profile's copy into root so the read below borrows root's grant
# (#100339). No-op in classic mode or once the profile is clean.
auth_mod.heal_forked_single_use_oauth_grants(provider)
raw_entries = read_credential_pool(provider)
disk_ids = {
entry.get("id")
@@ -3678,18 +3827,41 @@ def load_pool(provider: str) -> CredentialPool:
# process missing a provider env var must not delete the persisted
# pool entry for every other process (#9331). File-backed singletons
# still prune when their backing file is gone.
changed |= _prune_stale_seeded_entries(
entries,
singleton_sources | env_sources,
prune_env_sources=False,
borrowing_root_grant = (
provider in SINGLE_USE_REFRESH_POOL_PROVIDERS
and bool(disk_ids)
and not _profile_owns_pool_provider(provider)
)
if borrowing_root_grant:
# Rows read through the global-root fallback are seeded from the
# ROOT's singleton files, which this profile cannot see; pruning
# them as "backing file gone" would hide (and, via write-through,
# delete) the shared grant. The root's own load_pool() prunes.
borrowed = [e for e in entries if e.id in disk_ids]
others = [e for e in entries if e.id not in disk_ids]
changed |= _prune_stale_seeded_entries(
others, singleton_sources | env_sources, prune_env_sources=False,
)
entries[:] = borrowed + others
else:
changed |= _prune_stale_seeded_entries(
entries,
singleton_sources | env_sources,
prune_env_sources=False,
)
changed |= _normalize_pool_priorities(provider, entries)
if changed:
new_ids = {entry.id for entry in entries}
write_credential_pool(
persist_pool_entries(
provider,
[entry.to_dict() for entry in sorted(entries, key=lambda item: item.priority)],
removed_ids=disk_ids - new_ids,
)
return CredentialPool(provider, entries)
pool = CredentialPool(provider, entries)
# Remember which rows are the root's grant (borrowed via fallback) so a
# later ``add_entry`` in this profile can leave them out of the profile's
# own store (#100339).
if provider in SINGLE_USE_REFRESH_POOL_PROVIDERS and not _profile_owns_pool_provider(provider):
pool._borrowed_root_ids = set(disk_ids)
return pool
+3 -8
View File
@@ -252,15 +252,10 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool:
if not base_url:
return False
try:
from hermes_cli.models import _is_model_free, _pricing_cache
from hermes_cli.models import _is_model_free, peek_cached_pricing
# Mirror get_pricing_for_provider's key normalization: the agent's
# Nous base_url is /v1-suffixed (https://inference-api.nousresearch.com/v1)
# but the picker keys _pricing_cache on the pre-/v1 root.
key = base_url.rstrip("/")
if key.endswith("/v1"):
key = key[:-3].rstrip("/")
pricing = _pricing_cache.get(key)
# peek_cached_pricing owns the /v1-suffix and auth-state key details.
pricing = peek_cached_pricing(base_url)
if not pricing:
return False
return _is_model_free(model, pricing)
+8 -8
View File
@@ -462,7 +462,7 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) -
"image_generate": "prompt", "text_to_speech": "text",
"vision_analyze": "question",
"skill_view": "name", "skills_list": "category",
"cronjob": "action",
"cronjob_manage": "action",
"execute_code": "code", "browser_exec": "code", "delegate_task": "goal",
"clarify": "question", "skill_manage": "name",
}
@@ -496,7 +496,7 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) -
preview = _oneline(str(goal))
return _truncate_preview(preview, max_len) if preview else None
if tool_name == "process":
if tool_name == "process_manage":
action = args.get("action", "")
sid = args.get("session_id", "")
data = args.get("data", "")
@@ -511,7 +511,7 @@ def build_tool_preview(tool_name: str, args: dict, max_len: int | None = None) -
parts = [p for p in parts if p]
return " ".join(parts) if parts else None
if tool_name == "todo":
if tool_name == "todo_list":
todos_arg = args.get("todos")
merge = args.get("merge", False)
if todos_arg is None:
@@ -657,10 +657,10 @@ _TOOL_VERBS: dict[str, str] = {
"skills_list": "Listing skills",
"skill_manage": "Updating skill",
"delegate_task": "Delegating",
"cronjob": "Scheduling",
"cronjob_manage": "Scheduling",
"clarify": "Asking",
"memory": "Updating memory",
"todo": "Updating tasks",
"todo_list": "Updating tasks",
}
# Verbs that read better without the raw argument preview appended.
@@ -1433,7 +1433,7 @@ def _get_cute_tool_message(
return _wrap(f"┊ 📄 fetch pages {dur}")
if tool_name == "terminal":
return _wrap(f"┊ 💻 $ {_trunc(build_tool_preview(tool_name, args) or args.get('command', ''), 42)} {dur}")
if tool_name == "process":
if tool_name == "process_manage":
action = args.get("action", "?")
sid = args.get("session_id", "")[:12]
labels = {"list": "ls processes", "poll": f"poll {sid}", "log": f"log {sid}",
@@ -1473,7 +1473,7 @@ def _get_cute_tool_message(
return _wrap(f"┊ 🖼️ images extracting {dur}")
if tool_name == "browser_vision":
return _wrap(f"┊ 👁️ vision analyzing page {dur}")
if tool_name == "todo":
if tool_name == "todo_list":
todos_arg = args.get("todos")
merge = args.get("merge", False)
# Parse result for completion progress
@@ -1532,7 +1532,7 @@ def _get_cute_tool_message(
return _wrap(f"┊ 👁️ vision {_trunc(args.get('question', ''), 30)} {dur}")
if tool_name == "send_message":
return _wrap(f"┊ 📨 send {args.get('target', '?')}: \"{_trunc(args.get('message', ''), 25)}\" {dur}")
if tool_name == "cronjob":
if tool_name == "cronjob_manage":
action = args.get("action", "?")
if action == "create":
skills = args.get("skills") or ([] if not args.get("skill") else [args.get("skill")])
+63
View File
@@ -0,0 +1,63 @@
"""Bounded fast-mode windows (``/fast auto`` and ``/fast cold``).
``agent.service_tier`` is ``None`` (normal), ``"priority"`` (static fast),
``"auto"`` or ``"cold"``. The static value is pinned into
``agent.request_overrides`` at agent build time; the two bounded modes
instead open a wall-clock window at each user-turn boundary and layer the
provider's fast override onto the request kwargs only while it is open:
- ``auto`` — every user turn opens a window of ``agent.fast_auto_seconds``.
- ``cold`` — only the first turn of a session (no prior history) opens it.
Only per-request params (``service_tier`` / ``speed``) vary between requests;
the system prompt, tools, and messages are untouched, so the prompt cache is
preserved across the window boundary.
"""
from __future__ import annotations
import time
from typing import Any
BOUNDED_MODES = frozenset({"auto", "cold"})
DEFAULT_WINDOW_SECONDS = 60
def begin_turn(agent: Any, conversation_history: Any) -> None:
"""Open (or refuse) the fast window at a user-turn boundary."""
mode = getattr(agent, "service_tier", None)
agent._fast_until = 0.0
if mode not in BOUNDED_MODES:
return
if mode == "cold" and any(
isinstance(m, dict) and m.get("role") in ("user", "assistant", "tool")
for m in (conversation_history or ())
):
return
try:
window = float(getattr(agent, "fast_auto_seconds", DEFAULT_WINDOW_SECONDS))
except (TypeError, ValueError):
window = DEFAULT_WINDOW_SECONDS
agent._fast_until = time.monotonic() + max(window, 0.0)
def effective_request_overrides(agent: Any) -> dict[str, Any]:
"""``agent.request_overrides`` plus the fast override while the window is open."""
overrides = dict(getattr(agent, "request_overrides", None) or {})
if getattr(agent, "service_tier", None) not in BOUNDED_MODES:
return overrides
if time.monotonic() >= getattr(agent, "_fast_until", 0.0):
return overrides
from hermes_cli.models import resolve_fast_mode_overrides
base_url = getattr(agent, "base_url", None)
if getattr(agent, "api_mode", None) == "anthropic_messages":
base_url = getattr(agent, "_anthropic_base_url", None) or base_url
fast = resolve_fast_mode_overrides(
getattr(agent, "model", None),
provider=getattr(agent, "provider", None),
base_url=base_url,
)
if fast:
overrides.update(fast)
return overrides
+44 -3
View File
@@ -519,6 +519,28 @@ def _lookup_supports_vision(
return override
if not provider or not model:
return None
# Managed local runtime: the server that would receive the image is
# the authority on whether it can see (its /props reports modalities
# when a vision projector is loaded; the catalog covers staged-but-
# unloaded models). Cloud catalogs have never heard of a local GGUF,
# so without this answer every local model reads as text-only and
# images detour to a cloud auxiliary — wrong twice for a local-first
# user (broken feature, and a screenshot leaving the machine).
try:
from hermes_cli.local_runtime.capabilities import (
is_managed_provider,
managed_model_supports_vision,
)
if is_managed_provider(provider, _resolve_inference_base_url(cfg, provider) or ""):
managed = managed_model_supports_vision(model)
if managed is not None:
return managed
except Exception as exc: # pragma: no cover - defensive
logger.debug("image_routing: managed-runtime caps lookup failed for %s:%s — %s",
provider, model, exc)
caps = None
try:
from agent.models_dev import get_model_capabilities
@@ -813,12 +835,31 @@ def _file_to_data_url(path: Path) -> Optional[str]:
logger.warning("image_routing: failed to read %s — %s", path, exc)
return None
mime = _guess_mime(path, raw=raw)
if mime not in _UNIVERSALLY_SUPPORTED_MIMES:
accepted = _UNIVERSALLY_SUPPORTED_MIMES
# The managed local server decodes fewer formats than cloud providers
# (no WebP — and a WebP part fails SILENTLY: the model never sees an
# image and confabulates a description). When the active main model is
# served by the managed runtime, narrow the accepted set so those
# formats transcode to PNG here instead of vanishing server-side.
try:
from agent.auxiliary_client import _runtime_main_value
from hermes_cli.local_runtime.capabilities import (
ACCEPTED_IMAGE_MIMES,
is_managed_provider,
)
if is_managed_provider(
str(_runtime_main_value("provider") or ""),
str(_runtime_main_value("base_url") or "")):
accepted = ACCEPTED_IMAGE_MIMES
except Exception: # noqa: BLE001 — best-effort narrowing only
pass
if mime not in accepted:
transcoded = _transcode_to_png(raw)
if transcoded is None:
logger.warning(
"image_routing: %s is %s which is not accepted by all major "
"vision providers and could not be transcoded to PNG; "
"image_routing: %s is %s which is not accepted by the "
"active provider and could not be transcoded to PNG; "
"skipping this attachment.",
path, mime,
)
+122 -1
View File
@@ -370,6 +370,58 @@ def _save_model_metadata_disk_cache(data: Dict[str, Dict[str, Any]]) -> None:
except Exception as e:
logger.debug("Failed to save OpenRouter model metadata disk cache: %s", e)
def _get_endpoint_metadata_cache_path() -> Path:
"""On-disk memo of remote ``/models`` probes (see ``_endpoint_disk_cache_get``)."""
from hermes_constants import get_hermes_home
return get_hermes_home() / "cache" / "endpoint_model_metadata.json"
def _endpoint_disk_cache_get(normalized: str) -> Optional[Dict[str, Dict[str, Any]]]:
"""Return a still-fresh (``_ENDPOINT_MODEL_CACHE_TTL``) disk memo for one endpoint.
The in-memory endpoint cache only helps within a process. One-shot runs
(``hermes -q``, cron, every Bot Mode DM hop) start cold and re-probed the
live ``/models`` endpoint on every launch — 0.3–0.6s of pure network per
process on Nous, whose persistent context cache is bypassed by design so
the portal stays authoritative. This memo keeps that authority (same TTL
as the in-memory cache, so reconciliation still lands within 5 minutes)
while sharing the answer across processes. Local endpoints are never
memoized: their loaded context is transient (LM Studio reloads).
"""
try:
with _get_endpoint_metadata_cache_path().open("r", encoding="utf-8") as f:
data = json.load(f)
entry = data.get(normalized) if isinstance(data, dict) else None
if not isinstance(entry, dict):
return None
if (time.time() - float(entry.get("at", 0))) >= _ENDPOINT_MODEL_CACHE_TTL:
return None
models = entry.get("models")
return models if isinstance(models, dict) else None
except Exception:
return None
def _endpoint_disk_cache_put(normalized: str, cache: Dict[str, Dict[str, Any]]) -> None:
"""Memoize a successful remote ``/models`` probe; expired siblings are dropped."""
try:
path = _get_endpoint_metadata_cache_path()
data: Dict[str, Any] = {}
if path.exists():
with path.open("r", encoding="utf-8") as f:
loaded = json.load(f)
if isinstance(loaded, dict):
now = time.time()
data = {
k: v for k, v in loaded.items()
if isinstance(v, dict) and (now - float(v.get("at", 0))) < _ENDPOINT_MODEL_CACHE_TTL
}
data[normalized] = {"at": time.time(), "models": cache}
atomic_json_write(path, data, indent=0, separators=(",", ":"))
except Exception as e:
logger.debug("Failed to save endpoint model metadata disk cache: %s", e)
# Descending tiers for context length probing when the model is unknown.
# We start at 256K (covers GPT-5.x, many current large-context models) and
# step down on context-length errors until one works. Tier[0] is also the
@@ -1362,6 +1414,12 @@ def fetch_endpoint_model_metadata(
cached_at = _endpoint_model_metadata_cache_time.get(normalized, 0)
if cached is not None and (time.time() - cached_at) < _ENDPOINT_MODEL_CACHE_TTL:
return cached
if not is_local_endpoint(normalized):
memo = _endpoint_disk_cache_get(normalized)
if memo is not None:
_endpoint_model_metadata_cache[normalized] = memo
_endpoint_model_metadata_cache_time[normalized] = time.time()
return memo
# Blackholed endpoint: every candidate below would spend its full 5s
# connect budget. Returned empty rather than cached, so the endpoint is
@@ -1502,11 +1560,42 @@ def fetch_endpoint_model_metadata(
model_alias = props.get("model_alias", "")
if n_ctx and model_alias and model_alias in cache:
cache[model_alias]["context_length"] = n_ctx
else:
# Router mode: bare /props 400s and telemetry is
# per-child (?model=). Enumerate children via the
# native /models (carries status) and read each
# LOADED child's granted window — the value the
# context policy actually granted, which the meter
# and compressor must follow. Unloaded children are
# skipped: probing them could trigger an autoload.
native = requests.get(base + "/models", headers=headers, timeout=5, verify=_verify)
if native.ok:
children = (native.json() or {}).get("data", [])
for child in children[:16]:
if not isinstance(child, dict):
continue
child_id = child.get("id")
status = (child.get("status") or {}).get("value")
if not child_id or child_id not in cache or status not in ("loaded", "ready"):
continue
pr = requests.get(
base + "/v1/props", params={"model": child_id},
headers=headers, timeout=5, verify=_verify)
if not pr.ok:
pr = requests.get(
base + "/props", params={"model": child_id},
headers=headers, timeout=5, verify=_verify)
if pr.ok:
child_ctx = (pr.json().get("default_generation_settings") or {}).get("n_ctx")
if child_ctx:
cache[child_id]["context_length"] = child_ctx
except Exception:
pass
_endpoint_model_metadata_cache[normalized] = cache
_endpoint_model_metadata_cache_time[normalized] = time.time()
if cache and not is_local_endpoint(normalized):
_endpoint_disk_cache_put(normalized, cache)
return cache
except Exception as exc:
last_error = exc
@@ -2375,6 +2464,27 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
return int(ctx)
break
# llama.cpp: /props reports default_generation_settings.n_ctx —
# the RUNTIME window the server grants. Critically, the router
# answers this (from its preset) even for a model that is not
# currently loaded, while /v1/models reports meta=null until
# load. Without this probe, resolving a lazily-loaded model at
# session start finds no metadata and falls through to the
# name-pattern defaults, where a family catch-all (e.g. "qwen"
# = 131072) misreports a server launched at 262144.
if server_type == "llamacpp":
for props_path in (f"/props?model={model}", "/props"):
try:
resp = client.get(f"{server_url}{props_path}")
except httpx.HTTPError:
break
if resp.status_code != 200:
continue
n_ctx = (resp.json().get("default_generation_settings")
or {}).get("n_ctx")
if isinstance(n_ctx, (int, float)) and n_ctx:
return int(n_ctx)
# LM Studio / vLLM / llama.cpp / Anthropic-compat proxies:
# try /v1/models/{model}
resp = client.get(f"{server_url}/v1/models/{model}")
@@ -3908,6 +4018,8 @@ def capture_usage_anchor(
def anchored_context_tokens(
messages: List[Dict[str, Any]],
anchor: Optional[Dict[str, Any]],
*,
charge_stale_thinking: bool = True,
) -> Optional[int]:
"""Context size anchored on the last provider-reported usage.
@@ -3917,6 +4029,13 @@ def anchored_context_tokens(
estimation). The assistant reply produced by the anchored response
(first appended message after the base) is skipped: its cost is already
counted exactly by ``completion_tokens``.
``charge_stale_thinking`` is forwarded to the delta estimate — pass
``False`` to exclude transient ``reasoning``/``reasoning_content`` text
on all but the newest assistant message in the delta (the durable-
transcript view used by display surfaces; see the turn-base anchor in
``agent/conversation_loop.py``). Default ``True`` preserves the
conservative full charge for request-size callers.
"""
if not isinstance(anchor, dict) or not isinstance(messages, list):
return None
@@ -3938,7 +4057,9 @@ def anchored_context_tokens(
# completion_tokens above.
delta = delta[1:]
if delta:
total += estimate_messages_tokens_rough(delta)
total += estimate_messages_tokens_rough(
delta, charge_stale_thinking=charge_stale_thinking
)
return total
+37 -3
View File
@@ -98,8 +98,12 @@ _TOOL_SCOPED_EVENTS = {"pre_tool_call", "post_tool_call"}
# kwargs promoted to top-level payload keys (mirrors shell hooks wire).
_TOP_LEVEL_PAYLOAD_KEYS = {"tool_name", "args", "session_id", "parent_session_id"}
# (event, url) pairs already wired to the plugin manager in this process.
_registered: Set[Tuple[str, str]] = set()
# (home, event, url) triples already wired to the plugin manager in this
# process. Home is part of the key so a multiplexed gateway's secondary
# profiles — each with their own plugin manager (see
# hermes_cli.plugins.get_plugin_manager) — can register identical webhook
# targets without the first profile's registration shadowing the rest.
_registered: Set[Tuple[str, str, str]] = set()
_registered_lock = threading.Lock()
_delivery_queue: "queue.Queue[Optional[Dict[str, Any]]]" = queue.Queue(
@@ -180,15 +184,17 @@ def register_from_config(cfg: Optional[Dict[str, Any]]) -> List[WebhookTarget]:
return []
from hermes_cli.plugins import get_plugin_manager
from hermes_constants import get_hermes_home
manager = get_plugin_manager()
home_key = str(get_hermes_home().expanduser().resolve())
registered: List[WebhookTarget] = []
with _registered_lock:
for target in targets:
wired_any = False
for event in target.events:
key = (event, target.url)
key = (home_key, event, target.url)
if key in _registered:
continue
manager._hooks.setdefault(event, []).append(
@@ -231,6 +237,29 @@ def flush(timeout: float = 5.0) -> bool:
return _delivery_queue.unfinished_tasks == 0
def re_register_config_hooks() -> None:
"""Re-register outbound webhooks from config after a plugin force-reload.
Mirrors ``agent.shell_hooks.re_register_config_hooks``: config-owned
outbound-webhook callbacks live in the same ``_hooks`` dict that
``PluginManager.discover_and_load(force=True)`` clears via ``unload()``,
so without this the force-reloaded profile's outbound webhooks go
silently inert (#92682 review). Only the current home's idempotence
keys are cleared so a force-reload in one profile cannot invalidate
another profile's still-live registration.
"""
from hermes_cli.config import load_config
from hermes_constants import get_hermes_home
home_key = str(get_hermes_home().expanduser().resolve())
with _registered_lock:
_registered.difference_update(
{key for key in _registered if key[0] == home_key}
)
register_from_config(load_config())
def reset_for_tests() -> None:
"""Clear the idempotence set and drain the queue. Test-only helper."""
with _registered_lock:
@@ -416,8 +445,13 @@ def _serialize_payload(
cwd = str(Path.cwd())
except OSError:
cwd = ""
# Resolved at fire time from the bound home so a multiplexed gateway's
# receivers can tell which profile emitted the event (#92674).
from hermes_cli.profiles import get_active_profile_name
payload = {
"hook_event_name": event,
"profile": get_active_profile_name(),
"tool_name": kwargs.get("tool_name"),
"tool_input": kwargs.get("args") if isinstance(kwargs.get("args"), dict) else None,
"session_id": kwargs.get("session_id") or kwargs.get("parent_session_id") or "",
+48
View File
@@ -269,6 +269,49 @@ def _enable_happy_eyeballs(transport) -> None:
pool._network_backend = _HappyEyeballsSyncBackend()
def enable_happy_eyeballs_on_client(client) -> None:
"""Install the sync racing backend on every direct transport of a client.
Covers a ready-built ``httpx.Client`` (its default transport plus any
mounts), for callers that construct clients inline instead of going
through :func:`build_keepalive_http_client` — e.g. the Codex OAuth token
refresh / device-login / usage-probe clients in ``hermes_cli.auth``.
Proxy-backed transports (``httpcore.HTTPProxy`` / SOCKS pools) are left
untouched: with a proxy in play the TCP connect goes to the proxy host,
which is out of scope for the direct-transport racing added in #94388.
Async clients are also left untouched — httpcore's async backend already
performs RFC 8305 racing natively via
``anyio.connect_tcp(happy_eyeballs_delay=0.25)``.
Best-effort and hasattr-guarded like ``_enable_happy_eyeballs``; on an
incompatible httpx/httpcore this silently keeps the default backend.
"""
try:
import httpcore
proxy_pool_types = tuple(
t
for t in (
getattr(httpcore, "HTTPProxy", None),
getattr(httpcore, "SOCKSProxy", None),
)
if t is not None
)
except Exception:
return
transports = [getattr(client, "_transport", None)]
transports.extend((getattr(client, "_mounts", None) or {}).values())
for transport in transports:
pool = getattr(transport, "_pool", None)
if pool is None or not hasattr(pool, "_network_backend"):
continue
if proxy_pool_types and isinstance(pool, proxy_pool_types):
continue
pool._network_backend = _HappyEyeballsSyncBackend()
def _load_openai_cls() -> type:
"""Import and cache ``openai.OpenAI``."""
global _OPENAI_CLS_CACHE
@@ -421,6 +464,10 @@ def build_keepalive_http_client(
if proxy is None:
http_transport = transport_cls(verify=verify)
https_transport = transport_cls(verify=verify)
# Async transports need no explicit racing: httpcore's anyio
# backend already implements RFC 8305 natively
# (``anyio.connect_tcp(happy_eyeballs_delay=0.25)``), covered by
# tests/agent/test_codex_happy_eyeballs.py.
if not async_mode and _uses_codex_cloud_transport(base_url):
_enable_happy_eyeballs(http_transport)
_enable_happy_eyeballs(https_transport)
@@ -459,4 +506,5 @@ __all__ = [
"_get_proxy_from_env",
"_get_proxy_for_base_url",
"build_keepalive_http_client",
"enable_happy_eyeballs_on_client",
]
+20 -2
View File
@@ -1173,6 +1173,24 @@ _WINDOWS_BASH_SHELL_HINT = (
)
def _tenv_read(name: str, default: str = "") -> str:
"""Scope-aware TERMINAL_* read (tools.terminal_scope.terminal_env).
The per-turn terminal scope installed by the multiplexing gateway carries
the active profile's terminal settings; a raw os.getenv would read a value
a previous profile's turn pinned into the process env.
Only an import failure falls back: an active refusal scope must raise —
swapping it for the ambient process value would defeat the fail-closed
boundary.
"""
try:
from tools.terminal_scope import terminal_env
except ImportError:
return os.getenv(name, default)
return terminal_env(name, default)
def _probe_remote_backend(env_type: str) -> str | None:
"""Run a tiny introspection command inside the active terminal backend.
@@ -1181,7 +1199,7 @@ def _probe_remote_backend(env_type: str) -> str | None:
per process. Used only for non-local backends where the agent's tools
operate on a different machine than the host Hermes runs on.
"""
cwd_hint = os.getenv("TERMINAL_CWD", "")
cwd_hint = _tenv_read("TERMINAL_CWD", "")
cache_key = (env_type, cwd_hint)
cached = _BACKEND_PROBE_CACHE.get(cache_key)
if cached is not None:
@@ -1330,7 +1348,7 @@ def build_environment_hints() -> str:
hints: list[str] = []
backend = (os.getenv("TERMINAL_ENV") or "local").strip().lower()
backend = (_tenv_read("TERMINAL_ENV") or "local").strip().lower()
is_remote_backend = backend in _REMOTE_TERMINAL_BACKENDS or _plugin_backend_is_remote(backend)
if not is_remote_backend:
+52
View File
@@ -270,6 +270,16 @@ _KEY_KEYWORD_RE = re.compile(
re.IGNORECASE,
)
# Key names that are credential-specific even when their values are short or
# human-readable. Bare ``token`` / ``key`` are intentionally absent: those
# words also describe model limits, tensor names, cache keys, and other public
# technical values. Their assignments are gated on value shape below.
_STRONG_KEY_KEYWORD_RE = re.compile(
r"(?:api|auth|access|refresh|session|id|bearer)[ _.\\-]?(?:key|token)"
r"|key[ _.\\-]?material|secret|passwd|password|pass|pw|credential|auth|bearer",
re.IGNORECASE,
)
def _is_word_start(s: str, i: int) -> bool:
"""True if position ``i`` in ``s`` begins a word (not mid-word)."""
@@ -326,6 +336,42 @@ def _key_has_secret_keyword(key: str) -> bool:
return True
return False
def _key_has_strong_secret_keyword(key: str) -> bool:
"""Return whether ``key`` names an unambiguously credential-bearing field."""
for match in _STRONG_KEY_KEYWORD_RE.finditer(key):
if _is_word_start(key, match.start()) and _is_word_end(key, match.end()):
return True
return False
def _looks_like_opaque_credential(value: str) -> bool:
"""Return whether an ambiguous token/key value has credential-like shape.
Known vendor prefixes and JWTs have dedicated redactors. This catches the
remaining opaque family without treating short technical scalars such as
``CPU``, ``local``, or training captions as secrets merely because their
key contains ``token`` or ``key``.
"""
if value == "***" or value.startswith("«redacted:"):
return True
if len(value) >= 16 and re.fullmatch(r"[A-Fa-f0-9]+", value):
return True
if len(value) >= 20 and re.fullmatch(r"[A-Za-z0-9_./+=-]+", value):
return True
if len(value) < 12:
return False
classes = sum(
bool(re.search(pattern, value))
for pattern in (r"[a-z]", r"[A-Z]", r"[0-9]")
)
return classes >= 2
def _assignment_value_requires_redaction(key: str, value: str) -> bool:
"""Apply value-aware gating to key-name-only assignment matches."""
return _key_has_strong_secret_keyword(key) or _looks_like_opaque_credential(value)
# JSON field patterns: "apiKey": "value", "token": "value", etc.
_JSON_KEY_NAMES = r"(?:api_?[Kk]ey|token|secret|password|access_token|refresh_token|auth_token|bearer|secret_value|raw_secret|secret_input|key_material)"
_JSON_FIELD_RE = re.compile(
@@ -870,6 +916,8 @@ def redact_sensitive_text(
# embedded matching inside the helper.
if not _key_has_secret_keyword(name):
return m.group(0)
if not _assignment_value_requires_redaction(name, value):
return m.group(0)
return f"{name}={quote}{_mask_token(value)}{quote}"
text = _ENV_ASSIGN_RE.sub(_redact_env, text)
# Lowercase env names (``openai_key=…``). Skip URLs — the query
@@ -905,6 +953,8 @@ def redact_sensitive_text(
# not a leaked secret value.
if _ENV_LOOKUP_VALUE_RE.match(value):
return m.group(0)
if not _assignment_value_requires_redaction(key, value):
return m.group(0)
return f'{key}: "{_mask_token(value)}"'
text = _JSON_FIELD_RE.sub(_redact_json, text)
@@ -924,6 +974,8 @@ def redact_sensitive_text(
# document text, not credentials (nearai/ironclaw#6129).
if not _key_has_secret_keyword(key):
return m.group(0)
if not _assignment_value_requires_redaction(key, value):
return m.group(0)
return f"{key}{sep}{_mask_token(value)}"
text = _YAML_ASSIGN_RE.sub(_redact_yaml, text)
+291
View File
@@ -0,0 +1,291 @@
"""Idle deferral for background reviews on the managed local runtime.
The post-turn review fork replays the whole conversation on the review
runtime. On a cloud provider that costs seconds and runs concurrently
with whatever the user does next. When the review runtime IS the managed
llama-server, the same fork monopolizes the GPU the user's next prompt
needs, for minutes — and the next live turn cancels it, so an active
session tends to pay the decode cost AND lose the learning.
This module keeps the decision to learn exactly where it was (turn end,
nudge intervals, full-strength model, full transcript) and moves only
the execution moment: reviews bound for the managed local endpoint are
queued and dispatched when the machine is quiet. Everything else runs
immediately, as before.
Policy (auxiliary.background_review.defer):
auto (default) — defer exactly when the resolved review runtime
targets the managed local server.
never — old behavior everywhere.
Explicit /refine (focus set) never defers: an explicit ask runs now,
matching its bypass of the enabled gate.
Queue semantics:
- One slot per session, newest snapshot wins. A review replays the whole
conversation, so a newer snapshot strictly supersedes an older one —
coalescing is deduplication, not loss.
- Preempted (cancelled-by-live-turn) reviews are requeued by the spawn
wrapper observing the run token's cancel flag, not killed-and-forgotten.
- Aged-out events (defer_max_age_s, default 30 min) dispatch regardless
of idleness — deferral may delay learning, never lose it.
- In-memory, best-effort: dropped on process exit, the same durability
contract the immediate daemon-thread fork always had.
Idle truth comes from the supervisor's /slots (machine-level: it sees
every client of the managed server, including other Hermes profiles) and
must hold for a settle window so a review is not launched into the gap
between two quick prompts. Local in-process turn liveness is tracked via
note_turn_started/note_turn_finished from run_conversation.
"""
from __future__ import annotations
import json
import logging
import threading
import time
import urllib.request
from typing import Any, Callable, Dict, List, Optional
logger = logging.getLogger(__name__)
# Sustained-quiet window before dispatch. Long enough that "typed two
# prompts back to back" does not look idle; short enough that walking
# away for coffee runs the queue.
_IDLE_SETTLE_S = 15.0
# Poll cadence while the queue is non-empty. The thread parks when empty.
_POLL_INTERVAL_S = 5.0
# Age at which a queued review dispatches regardless of idleness.
_MAX_AGE_DEFAULT_S = 30.0 * 60.0
def defer_mode(task_cfg: Optional[Dict[str, Any]]) -> str:
"""'auto' (default) or 'never' from auxiliary.background_review.defer."""
raw = str((task_cfg or {}).get("defer", "auto")).strip().lower()
return raw if raw in ("auto", "never") else "auto"
def defer_max_age_s(task_cfg: Optional[Dict[str, Any]]) -> float:
raw = (task_cfg or {}).get("defer_max_age_s", _MAX_AGE_DEFAULT_S)
try:
value = float(raw)
except (TypeError, ValueError):
return _MAX_AGE_DEFAULT_S
return value if value > 0 else _MAX_AGE_DEFAULT_S
def review_targets_managed_local(agent: Any,
task_cfg: Optional[Dict[str, Any]]) -> bool:
"""Would this review fork decode on the llama-server WE manage?
Resolves the review runtime the same way the fork itself will and
exact-matches its netloc against the supervisor state file — the
matcher that cannot false-positive on external local servers. Any
failure reads False: immediate spawn is always the safe default.
Order matters: the netloc probe (one TTL-cached state-file read)
runs FIRST, so machines with no managed server — every cloud-only
install — return False without resolving the review runtime at all.
This wrapper runs on the turn's tail; runtime resolution belongs on
that path only when a managed server actually exists.
"""
try:
from agent.auxiliary_client import (
_is_managed_local_endpoint,
_managed_local_netloc,
)
if not _managed_local_netloc():
return False
from agent.background_review import _resolve_review_runtime
runtime = _resolve_review_runtime(agent, task_cfg)
return _is_managed_local_endpoint(runtime.get("base_url"))
except Exception: # noqa: BLE001
return False
class _PendingReview:
__slots__ = ("agent", "kwargs", "enqueued_at", "session_key")
def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any]):
self.agent = agent
self.session_key = session_key
self.kwargs = kwargs
self.enqueued_at = time.monotonic()
class ReviewIdleQueue:
"""Session-coalescing queue + idle-gated dispatcher thread."""
def __init__(self) -> None:
self._lock = threading.Lock()
self._pending: Dict[str, _PendingReview] = {}
self._wake = threading.Event()
self._thread: Optional[threading.Thread] = None
self._live_turns = 0
self._quiet_since: Optional[float] = None
# Test seams — replaced by unit tests, never in production.
self._now: Callable[[], float] = time.monotonic
self._server_idle: Callable[[], bool] = _managed_server_idle
# ── turn liveness (this process) ────────────────────────────
def note_turn_started(self) -> None:
with self._lock:
self._live_turns += 1
self._quiet_since = None
def note_turn_finished(self) -> None:
with self._lock:
self._live_turns = max(0, self._live_turns - 1)
if self._live_turns == 0:
self._quiet_since = self._now()
self._wake.set()
# ── queue ────────────────────────────────────────────────────
def enqueue(self, agent: Any, session_key: str,
kwargs: Dict[str, Any]) -> None:
"""Add (or replace — newest snapshot wins) a session's pending review."""
with self._lock:
existing = self._pending.get(session_key)
item = _PendingReview(agent, session_key, kwargs)
# Stamp through the queue's clock (test seam); keep the ORIGINAL
# enqueue time on coalesce so a busy session cannot push its
# review's age-out forever.
item.enqueued_at = (existing.enqueued_at if existing is not None
else self._now())
self._pending[session_key] = item
self._ensure_thread()
self._wake.set()
logger.info("Background review deferred (session=%s, queued=%d)",
session_key[-12:], len(self._pending))
def pending_count(self) -> int:
with self._lock:
return len(self._pending)
# ── dispatcher ───────────────────────────────────────────────
def _ensure_thread(self) -> None:
with self._lock:
if self._thread is None or not self._thread.is_alive():
self._thread = threading.Thread(
target=self._run, daemon=True, name="bg-review-idle-queue")
self._thread.start()
def _quiet_for(self) -> float:
"""Seconds this process has been turn-free (0 while a turn runs)."""
with self._lock:
if self._live_turns > 0 or self._quiet_since is None:
return 0.0
return self._now() - self._quiet_since
def _pop_dispatchable(self) -> Optional[_PendingReview]:
"""Oldest aged-out item, else any item once quiet+idle hold."""
with self._lock:
if not self._pending:
return None
items = sorted(self._pending.values(),
key=lambda p: p.enqueued_at)
aged = [p for p in items
if self._now() - p.enqueued_at
>= defer_max_age_s(p.kwargs.get("task_cfg"))]
candidate = aged[0] if aged else None
if candidate is None:
if self._quiet_for() < _IDLE_SETTLE_S:
return None
if not self._server_idle():
return None
with self._lock:
if not self._pending:
return None
candidate = min(self._pending.values(),
key=lambda p: p.enqueued_at)
with self._lock:
return self._pending.pop(candidate.session_key, None)
def _run(self) -> None:
while True:
self._wake.wait()
with self._lock:
if not self._pending:
self._wake.clear()
continue
item = None
try:
item = self._pop_dispatchable()
if item is not None:
if not self._still_enabled(item):
logger.info(
"Deferred background review dropped: reviews "
"were disabled while it was queued (session=%s)",
item.session_key[-12:])
continue
logger.info(
"Dispatching deferred background review "
"(session=%s, waited=%.0fs, queued=%d)",
item.session_key[-12:],
self._now() - item.enqueued_at,
self.pending_count())
item.agent._spawn_background_review_now(**item.kwargs)
except Exception: # noqa: BLE001 — dispatcher must survive anything
logger.warning("Deferred review dispatch failed",
exc_info=True)
if item is None:
time.sleep(_POLL_INTERVAL_S)
@staticmethod
def _still_enabled(item: _PendingReview) -> bool:
"""Re-check the enabled gate at DISPATCH time.
The entry wrapper gates at enqueue time, but minutes may pass in
the queue — a user who sets background_review.enabled: false while
a review waits means it, and the dispatch must not resurrect it.
Fail-open like the gate itself (a broken config never silently
disables reviews)."""
try:
from agent.background_review import load_background_review_settings
enabled, _ = load_background_review_settings()
return enabled
except Exception: # noqa: BLE001
return True
def _managed_server_idle() -> bool:
"""Machine-level idle: no processing slot on any loaded model of the
managed router. Unreachable/no state file reads idle (nothing to
contend with). One /models + one /slots call per loaded model."""
try:
from hermes_cli.local_runtime.supervisor import state_path
state = json.loads(state_path().read_text(encoding="utf-8"))
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
key = str(state.get("api_key", ""))
if not base:
return True
headers = {"Authorization": f"Bearer {key}"}
req = urllib.request.Request(f"{base}/models", headers=headers)
with urllib.request.urlopen(req, timeout=3) as r:
models = json.loads(r.read())
loaded = [m["id"] for m in models.get("data", [])
if (m.get("status") or {}).get("value") in ("loaded", "ready")]
from urllib.parse import quote
for mid in loaded:
req = urllib.request.Request(f"{base}/slots?model={quote(mid)}",
headers=headers)
with urllib.request.urlopen(req, timeout=3) as r:
slots = json.loads(r.read())
if any(s.get("is_processing") for s in slots
if isinstance(s, dict)):
return False
return True
except Exception: # noqa: BLE001
return True
# Module singleton — one queue per process, like the load-progress watcher.
QUEUE = ReviewIdleQueue()
+27 -2
View File
@@ -57,6 +57,31 @@ def _session_cwd_override() -> str:
return str(value).strip()
def _terminal_cwd_env() -> str:
"""Scope-aware TERMINAL_CWD read (tools.terminal_scope.terminal_env).
Under gateway multiplexing the per-turn terminal scope carries the active
profile's cwd; the process-global env var may hold another profile's
value. Only an import failure falls back: an active refusal scope must
raise, not silently resolve the launch profile's cwd.
"""
try:
from tools.terminal_scope import terminal_env
except ImportError:
return os.environ.get("TERMINAL_CWD", "")
return terminal_env("TERMINAL_CWD", "")
def scope_terminal_cwd() -> str:
"""Public wrapper — the scope-aware TERMINAL_CWD value (may be empty).
Shared by agent_init / skill_utils / code_execution_tool so every cwd
consumer reads through the per-turn terminal scope under gateway
multiplexing instead of the process-global env var.
"""
return _terminal_cwd_env()
def resolve_agent_cwd() -> Path:
override = _session_cwd_override()
if override:
@@ -64,7 +89,7 @@ def resolve_agent_cwd() -> Path:
if p.is_dir():
return p
logger.warning("configured working directory does not exist: %s", override)
raw = os.environ.get("TERMINAL_CWD", "").strip()
raw = _terminal_cwd_env().strip()
if raw:
p = Path(raw).expanduser()
if p.is_dir():
@@ -90,7 +115,7 @@ def resolve_context_cwd() -> Path | None:
else:
return p
return None
raw = os.environ.get("TERMINAL_CWD", "").strip()
raw = _terminal_cwd_env().strip()
if raw:
p = Path(raw).expanduser()
if not p.is_dir():
+20 -6
View File
@@ -182,13 +182,17 @@ _BLOCKING_EVENTS = frozenset({"pre_tool_call"})
_STDERR_MESSAGE_LIMIT = 400
# (event, matcher, command) triples that have been wired to the plugin
# (home, event, matcher, command) tuples that have been wired to the plugin
# manager in the current process. Matcher is part of the key because
# the same script can legitimately register for different matchers under
# the same event (e.g. one entry per tool the user wants to gate).
# Second registration attempts for the exact same triple become no-ops
# the same event (e.g. one entry per tool the user wants to gate). Home is
# part of the key so a multiplexed gateway's secondary profiles — each with
# their own plugin manager (see hermes_cli.plugins.get_plugin_manager) — can
# register identical hook triples without the first profile's registration
# silently shadowing the rest.
# Second registration attempts for the exact same tuple become no-ops
# so the CLI and gateway can both call register_from_config() safely.
_registered: Set[Tuple[str, Optional[str], str]] = set()
_registered: Set[Tuple[str, str, Optional[str], str]] = set()
_registered_lock = threading.Lock()
# Intra-process lock for allowlist read-modify-write on platforms that
@@ -289,13 +293,14 @@ def register_from_config(
from hermes_cli.plugins import get_plugin_manager
manager = get_plugin_manager()
home_key = str(get_hermes_home().expanduser().resolve())
# Idempotence + allowlist read happen under the lock; the TTY
# prompt runs outside so other threads aren't parked on a blocking
# input(). Mutation re-takes the lock with a defensive idempotence
# re-check in case two callers ever race through the prompt.
for spec in specs:
key = (spec.event, spec.matcher, spec.command)
key = (home_key, spec.event, spec.matcher, spec.command)
with _registered_lock:
if key in _registered:
continue
@@ -349,11 +354,20 @@ def re_register_config_hooks() -> None:
are wired again (#60036 / PR #60267; tracking #64178 — salvaged from
PR #64188).
Only the idempotence keys for the *current* Hermes home are cleared —
``discover_and_load(force=True)`` only unloads the manager scoped to
that one home, so clearing every home's keys would make a force-reload
in profile A drop profile B's still-live registration from the ledger
and duplicate it on B's next registration call (#92682 review).
Commands already allowlisted stay allowlisted, so this never re-prompts
at a TTY for hooks the user previously approved.
"""
home_key = str(get_hermes_home().expanduser().resolve())
with _registered_lock:
_registered.clear()
_registered.difference_update(
{key for key in _registered if key[0] == home_key}
)
from hermes_cli.config import load_config
register_from_config(load_config())
+11 -7
View File
@@ -236,7 +236,7 @@ def _load_skill_payload(skill_identifier: str, task_id: str | None = None) -> tu
return None
try:
from tools.skills_tool import SKILLS_DIR, skill_view
from tools.skills_tool import _skills_dir, skill_view
from agent.skill_utils import normalize_skill_lookup_name
normalized = normalize_skill_lookup_name(raw_identifier)
@@ -262,7 +262,7 @@ def _load_skill_payload(skill_identifier: str, task_id: str | None = None) -> tu
skill_dir = Path(abs_skill_dir)
elif skill_path:
try:
skill_dir = SKILLS_DIR / Path(skill_path).parent
skill_dir = _skills_dir() / Path(skill_path).parent
except Exception:
skill_dir = None
@@ -317,7 +317,7 @@ def _build_skill_message(
session_id: str | None = None,
) -> str:
"""Format a loaded skill into a user/system message payload."""
from tools.skills_tool import SKILLS_DIR
from tools.skills_tool import _skills_dir
content = str(loaded_skill.get("content") or "")
@@ -386,7 +386,7 @@ def _build_skill_message(
if supporting and skill_dir:
try:
skill_view_target = str(skill_dir.relative_to(SKILLS_DIR))
skill_view_target = str(skill_dir.relative_to(_skills_dir()))
except ValueError:
# Skill is from an external dir — use the skill name instead
skill_view_target = skill_dir.name
@@ -441,7 +441,7 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
# each naming the same skill as its own incumbent (#74574).
commands: Dict[str, Dict[str, Any]] = {}
try:
from tools.skills_tool import SKILLS_DIR, _parse_frontmatter, skill_matches_platform, skill_matches_environment, _get_disabled_skill_names
from tools.skills_tool import _skills_dir, _parse_frontmatter, skill_matches_platform, skill_matches_environment, _get_disabled_skill_names
from agent.skill_utils import (
get_external_skills_dirs,
get_project_skills_dirs,
@@ -456,8 +456,12 @@ def scan_skill_commands() -> Dict[str, Dict[str, Any]]:
# Project dirs iterate through the quarantine chokepoint.
project_dirs = list(get_project_skills_dirs())
dirs_to_scan = list(project_dirs)
if SKILLS_DIR.exists():
dirs_to_scan.append(SKILLS_DIR)
# Resolve at call time: the import-time SKILLS_DIR is frozen to the
# launch home, so a multiplexed profile scope (set_hermes_home_override)
# would still scan the default profile's skills (#67277).
skills_dir = _skills_dir()
if skills_dir.exists():
dirs_to_scan.append(skills_dir)
dirs_to_scan.extend(get_external_skills_dirs())
for scan_dir in dirs_to_scan:
+10 -5
View File
@@ -755,7 +755,9 @@ def find_project_root(start: Optional[Path] = None) -> Optional[Path]:
"""
try:
if start is None:
env_cwd = os.environ.get("TERMINAL_CWD")
from agent.runtime_cwd import scope_terminal_cwd
env_cwd = scope_terminal_cwd()
start = Path(env_cwd) if env_cwd else Path.cwd()
cur = Path(start).resolve()
except OSError:
@@ -994,12 +996,15 @@ def normalize_skill_lookup_name(identifier: str) -> str:
# Look the primary skills root up on tools.skills_tool at CALL time
# (not via get_skills_dir()): callers and tests patch
# ``tools.skills_tool.SKILLS_DIR`` and skill_view() itself resolves
# against that module attribute, so normalization must agree with the
# exact root skill_view() will enforce. Import deferred to avoid a
# module cycle (tools.skills_tool imports agent.skill_utils).
# against ``_skills_dir()`` — which honors that patch and otherwise
# follows the live profile-scoped HERMES_HOME (the import-time
# SKILLS_DIR is frozen to the launch home, #67277) — so normalization
# must agree with the exact root skill_view() will enforce. Import
# deferred to avoid a module cycle (tools.skills_tool imports
# agent.skill_utils).
try:
from tools import skills_tool as _skills_tool
primary_root = Path(_skills_tool.SKILLS_DIR)
primary_root = _skills_tool._skills_dir()
except Exception:
primary_root = get_skills_dir()
+18 -10
View File
@@ -1151,6 +1151,10 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
parsed_calls = []
for tool_call in tool_calls:
function_name = tool_call.function.name
# Legacy tool-name aliases (2026-08 renames) — map BEFORE the
# agent-loop branches (todo_list etc. dispatch above the registry).
from model_tools import _LEGACY_TOOL_ALIASES as _lta
function_name = _lta.get(function_name, function_name)
function_args, malformed_args_result = _parse_tool_arguments(
tool_call.function.arguments
@@ -1192,9 +1196,9 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe
_underlying, _underlying_args, _err = _ts.resolve_underlying_call(function_args)
if not _err and _underlying:
if _underlying in _tool_search_scoped_names(agent):
# Probe-validate before unwrapping (ironclaw#5149):
# missing required args return the parameter schema
# instead of dispatching into an opaque failure.
# Validate before unwrapping: the generic bridge hides
# the concrete parameter schema from provider-native
# tool-call validation.
_probe_err = _ts.validate_deferred_call_args(_underlying, _underlying_args)
if _probe_err is not None:
_ts_scope_block = _probe_err
@@ -2007,6 +2011,10 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
break
function_name = tool_call.function.name
# Legacy tool-name aliases (2026-08 renames) — map BEFORE the
# agent-loop branches (todo_list etc. dispatch above the registry).
from model_tools import _LEGACY_TOOL_ALIASES as _lta
function_name = _lta.get(function_name, function_name)
function_args, malformed_args_result = _parse_tool_arguments(
tool_call.function.arguments
@@ -2048,9 +2056,9 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
_underlying, _underlying_args, _err = _ts.resolve_underlying_call(function_args)
if not _err and _underlying:
if _underlying in _tool_search_scoped_names(agent):
# Probe-validate before unwrapping (ironclaw#5149):
# missing required args return the parameter schema
# instead of dispatching into an opaque failure.
# Validate before unwrapping: the generic bridge hides
# the concrete parameter schema from provider-native
# tool-call validation.
_probe_err = _ts.validate_deferred_call_args(_underlying, _underlying_args)
if _probe_err is not None:
# This path wraps _block_msg in {"error": ...} —
@@ -2081,7 +2089,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
tool_start_time = time.time()
if function_name == "todo":
if function_name == "todo_list":
def _execute(next_args: dict) -> Any:
from tools.todo_tool import todo_tool as _todo_tool
return _todo_tool(
@@ -2101,7 +2109,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
))
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('todo', function_args, tool_duration, result=function_result)}")
agent._vprint(f" {_get_cute_tool_message_impl('todo_list', function_args, tool_duration, result=function_result)}")
elif function_name == "message_agent":
# Bot Mode teammate DM (tools/bot_mode_dm.py) — injected, not
# registered: only a canonical Bot Chat session carries the
@@ -2336,7 +2344,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('read_window_below', function_args, tool_duration, result=function_result)}")
elif function_name == "tour":
elif function_name == "gui_tour":
def _execute(next_args: dict) -> Any:
from tools.tour_tool import tour_tool as _tour_tool
return _tour_tool(
@@ -2362,7 +2370,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
))
tool_duration = time.time() - tool_start_time
if agent._should_emit_quiet_tool_messages():
agent._vprint(f" {_get_cute_tool_message_impl('tour', function_args, tool_duration, result=function_result)}")
agent._vprint(f" {_get_cute_tool_message_impl('gui_tour', function_args, tool_duration, result=function_result)}")
elif function_name == "setup_mcp":
def _execute(next_args: dict) -> Any:
from tools.setup_mcp_tool import setup_mcp_tool as _setup_mcp_tool
+147 -10
View File
@@ -24,6 +24,8 @@ IDEMPOTENT_TOOL_NAMES = frozenset(
"web_search",
"web_extract",
"session_search",
"skill_view",
"skills_list",
"browser_snapshot",
"browser_console",
"browser_get_images",
@@ -44,7 +46,7 @@ MUTATING_TOOL_NAMES = frozenset(
"execute_code",
"write_file",
"patch",
"todo",
"todo_list",
"memory",
"skill_manage",
"browser_click",
@@ -53,9 +55,9 @@ MUTATING_TOOL_NAMES = frozenset(
"browser_scroll",
"browser_navigate",
"send_message",
"cronjob",
"cronjob_manage",
"delegate_task",
"process",
"process_manage",
}
)
@@ -67,7 +69,7 @@ MUTATING_TOOL_NAMES = frozenset(
# unannotated.
STALL_GUARD_REPEATABLE_TOOLS = frozenset(
{
"process",
"process_manage",
}
)
@@ -98,6 +100,49 @@ IDENTICAL_RESULT_STUB_MIN_CHARS = 512
_RESULT_STUB_ARGS_PREVIEW_CHARS = 120
# Tools whose "failure" is a normal, informative outcome of legitimate work:
# a red test run, a grep with no matches, a failing build during a fix loop, a
# page that times out. Hard stops never fire on these from failure counts of
# DIFFERENT commands (same_tool_failure) — only an exact-args replay with NO
# intervening change, or an identical-result streak, can halt them.
FAILURE_TOLERANT_TOOL_NAMES = frozenset(
{
"terminal",
"execute_code",
"process_manage",
"process",
"browser_navigate",
"web_extract",
}
)
# A landed mutation between two attempts means the retry is a NEW experiment
# (edit -> re-run) rather than a replay. A successful call to one of these
# marks progress for every failing signature still being counted this turn.
PROGRESS_RESET_TOOL_NAMES = frozenset(
{
"write_file",
"patch",
"terminal",
"execute_code",
"browser_click",
"browser_type",
"browser_press",
"browser_navigate",
"process_manage",
"process",
"delegate_task",
"send_message",
"cronjob",
"cronjob_manage",
"todo",
"todo_list",
"memory",
"skill_manage",
}
)
def is_stall_guard_repeatable(tool_name: str) -> bool:
"""Whether a tool is exempt from the identical-call loop notice."""
if tool_name in STALL_GUARD_REPEATABLE_TOOLS:
@@ -110,12 +155,14 @@ class ToolCallGuardrailConfig:
"""Thresholds for per-turn tool-call loop detection.
Warnings are enabled by default and never prevent tool execution. Hard stops
are explicit opt-in so interactive CLI/TUI sessions get a gentle nudge unless
the user enables circuit-breaker behavior in config.yaml.
stay opt-in for interactive CLI/TUI/Desktop/ACP sessions, but default on for
non-interactive gateway/cron platforms where nobody is present to interrupt
a model that ignores loop warnings.
"""
warnings_enabled: bool = True
hard_stop_enabled: bool = False
non_interactive_hard_stop_enabled: bool = True
exact_failure_warn_after: int = 2
exact_failure_block_after: int = 5
same_tool_failure_warn_after: int = 3
@@ -127,10 +174,15 @@ class ToolCallGuardrailConfig:
loop_caps: "LoopCapConfig" = field(default_factory=lambda: LoopCapConfig())
@classmethod
def from_mapping(cls, data: Mapping[str, Any] | None) -> "ToolCallGuardrailConfig":
def from_mapping(
cls,
data: Mapping[str, Any] | None,
*,
platform: str | None = None,
) -> "ToolCallGuardrailConfig":
"""Build config from the `tool_loop_guardrails` config.yaml section."""
if not isinstance(data, Mapping):
return cls()
data = {}
warn_after = data.get("warn_after")
if not isinstance(warn_after, Mapping):
@@ -140,9 +192,18 @@ class ToolCallGuardrailConfig:
hard_stop_after = {}
defaults = cls()
hard_stop_enabled = _as_bool(data.get("hard_stop_enabled"), defaults.hard_stop_enabled)
non_interactive_hard_stop_enabled = _as_bool(
data.get("non_interactive_hard_stop_enabled"),
defaults.non_interactive_hard_stop_enabled,
)
if _is_non_interactive_platform(platform) and non_interactive_hard_stop_enabled:
hard_stop_enabled = True
return cls(
warnings_enabled=_as_bool(data.get("warnings_enabled"), defaults.warnings_enabled),
hard_stop_enabled=_as_bool(data.get("hard_stop_enabled"), defaults.hard_stop_enabled),
hard_stop_enabled=hard_stop_enabled,
non_interactive_hard_stop_enabled=non_interactive_hard_stop_enabled,
exact_failure_warn_after=_positive_int(
warn_after.get("exact_failure", data.get("exact_failure_warn_after")),
defaults.exact_failure_warn_after,
@@ -218,6 +279,25 @@ class LoopCapConfig:
)
_INTERACTIVE_PLATFORMS = frozenset({"cli", "tui", "desktop", "acp"})
# Platforms that are not chat gateways but whose work is a bounded, supervised
# task loop: a subagent inherits its parent's budget and is stopped by the
# parent; api_server runs have a live client holding the request. Both do
# real edit -> re-run work, so they keep the interactive (warn-only) default.
_SUPERVISED_TASK_PLATFORMS = frozenset({"subagent", "api_server"})
def _is_non_interactive_platform(platform: str | None) -> bool:
"""Return true for gateway/cron sessions where tool loops are unattended."""
if not isinstance(platform, str) or not platform.strip():
return False
key = platform.strip().lower()
if key in _INTERACTIVE_PLATFORMS or key in _SUPERVISED_TASK_PLATFORMS:
return False
return True
@dataclass(frozen=True)
class IdenticalCallObservation:
"""Outcome of observing one completed tool call for the stall guards.
@@ -340,6 +420,8 @@ class ToolCallGuardrailController:
def reset_for_turn(self) -> None:
self._exact_failure_counts: dict[ToolCallSignature, int] = {}
self._same_tool_failure_counts: dict[str, int] = {}
# signature -> a mutating call succeeded since its last failure
self._progress_since_failure: dict[ToolCallSignature, bool] = {}
self._no_progress: dict[ToolCallSignature, tuple[str, int]] = {}
self._halt_decision: ToolGuardrailDecision | None = None
# Identical-call loop-breaker state (agent.stall_guards): tracks the
@@ -389,6 +471,10 @@ class ToolCallGuardrailController:
return ToolGuardrailDecision(tool_name=tool_name, signature=signature)
exact_count = self._exact_failure_counts.get(signature, 0)
if self._progress_since_failure.get(signature):
# Something landed since this call last failed — let it run; the
# streak restarts in after_call if it fails again.
exact_count = 0
if exact_count >= self.config.exact_failure_block_after:
decision = ToolGuardrailDecision(
action="block",
@@ -441,6 +527,12 @@ class ToolCallGuardrailController:
failed, _ = classify_tool_failure(tool_name, result)
if failed:
# An identical failing call is only a REPLAY if nothing landed in
# between. If any mutating call succeeded since the previous
# identical failure (edit -> re-run pytest, click -> re-snapshot),
# the retry is a new experiment: restart the exact-args streak.
if self._progress_since_failure.pop(signature, False):
self._exact_failure_counts.pop(signature, None)
exact_count = self._exact_failure_counts.get(signature, 0) + 1
self._exact_failure_counts[signature] = exact_count
self._no_progress.pop(signature, None)
@@ -448,7 +540,17 @@ class ToolCallGuardrailController:
same_count = self._same_tool_failure_counts.get(tool_name, 0) + 1
self._same_tool_failure_counts[tool_name] = same_count
if self.config.hard_stop_enabled and same_count >= self.config.same_tool_failure_halt_after:
# same_tool_failure counts DIFFERENT args on one tool. For tools
# whose non-zero exit is ordinary work output (terminal,
# execute_code, pollers) a run of distinct red commands is
# diagnosis, not a loop — warn, never halt. The exact-args replay
# path still applies to them.
same_tool_halt_eligible = tool_name not in FAILURE_TOLERANT_TOOL_NAMES
if (
self.config.hard_stop_enabled
and same_tool_halt_eligible
and same_count >= self.config.same_tool_failure_halt_after
):
decision = ToolGuardrailDecision(
action="halt",
code="same_tool_failure_halt",
@@ -492,6 +594,16 @@ class ToolCallGuardrailController:
self._exact_failure_counts.pop(signature, None)
self._same_tool_failure_counts.pop(tool_name, None)
# A successful mutation is progress for every failing signature still
# being counted this turn: the next identical retry runs against
# changed state, so it is a fresh attempt rather than a replay. Pure
# loops never mutate anything between attempts, so the replay detector
# keeps its teeth.
if tool_name in PROGRESS_RESET_TOOL_NAMES or file_mutation_result_landed(tool_name, result):
for sig in list(self._exact_failure_counts):
self._progress_since_failure[sig] = True
self._same_tool_failure_counts.clear()
if not self._is_idempotent(tool_name):
self._no_progress.pop(signature, None)
return ToolGuardrailDecision(tool_name=tool_name, signature=signature)
@@ -604,6 +716,31 @@ class ToolCallGuardrailController:
"Do not repeat it — change arguments, use a different tool, or "
"proceed with what you have.]"
)
# Hard-stop widening (#89069 / #100849 bundle): the per-turn
# no-progress BLOCK above only covers tools in idempotent_tools, so
# a model replaying the same successful `terminal`/`skill_view`
# call with a byte-identical result ran until the iteration budget.
# The consecutive-identical streak is tool-agnostic; when hard
# stops are enabled, halt at the same idempotent_no_progress
# threshold. Pollers stay exempt (an unchanged poll is progress).
if (
self.config.hard_stop_enabled
and count >= self.config.no_progress_block_after
and self._halt_decision is None
):
self._halt_decision = ToolGuardrailDecision(
action="halt",
code="identical_call_streak_halt",
message=(
f"Stopped {tool_name}: the same call with identical arguments "
f"returned the same result {count} times in a row. Stop "
"repeating it unchanged; use the result already provided or "
"change strategy."
),
tool_name=tool_name,
count=count,
signature=signature,
)
stub = None
if (
+10 -1
View File
@@ -767,9 +767,18 @@ class ChatCompletionsTransport(ProviderTransport):
extra_body["reasoning"] = gh_reasoning
else:
_effort = "medium"
_enabled = True
if reasoning_config and isinstance(reasoning_config, dict):
_effort = reasoning_config.get("effort", "medium") or "medium"
extra_body["reasoning"] = {"enabled": True, "effort": _effort}
# Honor an explicit "thinking off" (agent.reasoning_effort:
# none / the one-shot length-continuation override) the same
# way the provider-profile path does — never re-enable it.
if reasoning_config.get("enabled") is False or _effort == "none":
_enabled = False
if _enabled:
extra_body["reasoning"] = {"enabled": True, "effort": _effort}
else:
extra_body["reasoning"] = {"enabled": False, "effort": "none"}
if provider_name == "gemini":
raw_thinking_config = _build_gemini_thinking_config(model, reasoning_config)
+2 -2
View File
@@ -59,7 +59,7 @@ def _bounded_prompt_cache_key(value: Any) -> Optional[str]:
# A function literally named ``web_search`` collides with Grok's native
# server-side tool (incomplete hang or HTTP 400 duplicate names); this alias
# avoids that while still dispatching through Hermes's configured provider
# (Firecrawl / Exa / …). Mapped back to ``web_search`` in normalize_response.
# (Firecrawl / Tavily / …). Mapped back to ``web_search`` in normalize_response.
_XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search"
# OpenCode's /v1/responses endpoints (Zen and Go, including custom providers
@@ -661,7 +661,7 @@ class ResponsesApiTransport(ProviderTransport):
# fails): drop the client ``web_search`` function and declare
# xAI's built-in instead. 1:1 swap only when client ``web_search``
# was already present — never an additive grant.
# 2. **Client** (Firecrawl / Keenable / Exa / … configured or resolved):
# 2. **Client** (Firecrawl / Tavily / Exa / … configured or resolved):
# keep Hermes dispatch so ``web.backend`` / ``web.search_backend``
# is honored, but rename the wire tool to
# ``hermes_web_search`` so Grok cannot hijack the name. The alias
+16
View File
@@ -241,6 +241,22 @@ class TTSProvider(abc.ABC):
"if your backend supports it."
)
def warm(self) -> None:
"""Speech output was just turned on; pre-load so the first reply is hot.
Optional. Called from the TTS lease path (Desktop read-aloud / voice
conversation, ``/voice tts``) when this provider is the configured
``tts.provider`` — e.g. ask a local model server to load its model.
Best-effort: exceptions are logged at debug and ignored. Default: no-op.
"""
def release(self) -> None:
"""The last speech-output lease was released; free resident resources.
Optional counterpart of :meth:`warm` — e.g. tell a local model server
to unload. Best-effort; default: no-op.
"""
@property
def voice_compatible(self) -> bool:
"""Whether output is suitable for voice-bubble delivery.
+42 -7
View File
@@ -635,13 +635,18 @@ def build_turn_context(
# Between-turns MCP refresh: an MCP server that finished connecting since
# the previous turn (slow HTTP/OAuth servers routinely take 2-6s on a cold
# connect, missing the bounded startup wait) lands in THIS turn's tool
# snapshot. This is cache-safe by construction: it runs in the per-turn
# snapshot. Timing is cache-safe by construction: it runs in the per-turn
# prologue, before this turn's first API call assembles ``tools=``, so it
# only ever extends a fresh request prefix — it never mutates the cached
# prefix of an in-flight turn. No-op when no MCP servers are registered
# (the common case, gated by the cheap ``has_registered_mcp_tools`` check)
# or when the tool set is unchanged (``refresh_agent_mcp_tools`` diffs by
# name and leaves the snapshot untouched on no-change).
# never mutates the prefix of an in-flight turn. ``preserve_prefix`` makes
# the *content* cache-safe too (#100336): a plain rebuild re-derives the
# array from live availability, so a flapping ``check_fn`` silently drops a
# tool and a late arrival splices into sorted position — either one forks
# the tool block and re-prefills the whole history behind it, every turn it
# happens. With the flag the live order is authoritative and the array
# only ever grows. No-op when no MCP servers are registered (the common
# case, gated by the cheap ``has_registered_mcp_tools`` check) or when the
# tool set is unchanged (``refresh_agent_mcp_tools`` diffs by name and
# leaves the snapshot untouched on no-change).
try:
if not getattr(agent, "_skip_mcp_refresh", False):
# Import-cost gate: ``tools.mcp_tool`` pulls in the whole ``mcp``
@@ -656,7 +661,9 @@ def build_turn_context(
if "tools.mcp_tool" in _sys.modules:
from tools.mcp_tool import has_registered_mcp_tools, refresh_agent_mcp_tools
if has_registered_mcp_tools():
refresh_agent_mcp_tools(agent, quiet_mode=True)
refresh_agent_mcp_tools(
agent, quiet_mode=True, preserve_prefix=True,
)
except Exception:
logger.debug("between-turns MCP tool refresh skipped", exc_info=True)
@@ -1128,6 +1135,34 @@ def build_turn_context(
_compress_block_reason = _info(_preflight_tokens)[1]
except Exception:
_compress_block_reason = None
if _should_compress_now:
# Managed local runtime: growing the window beats compressing —
# the ladder's design order (same seam as the conversation
# loop's pre-API gate; see _maybe_grow_local_window there).
try:
from agent.conversation_loop import _maybe_grow_local_window
_grown = _maybe_grow_local_window(
agent, _compressor, _preflight_tokens
)
except Exception:
_grown = None
if _grown:
_compressor.update_model(
agent.model,
_grown,
base_url=getattr(agent, "base_url", "") or "",
api_key=getattr(agent, "api_key", "") or "",
provider=getattr(agent, "provider", "") or "",
api_mode=getattr(agent, "api_mode", "") or "",
)
agent._buffer_status(
f"📈 Context window grown to {_grown // 1024}K "
f"(local model; conversation continues uncompressed)"
)
_should_compress_now = _compressor.should_compress(
_preflight_tokens
)
if _should_compress_now:
_preflight_compressed = True
# Compression is actually running (block cleared / was never
+11
View File
@@ -126,6 +126,15 @@ def _drop_verification_continuation_scaffolding(messages) -> None:
]
def _clone_background_review_messages(messages):
"""Copy the review input without aliasing the live transcript."""
# Import lazily: conversation_loop imports this module during turn
# finalization, so a module-level import would create a cycle.
from agent.conversation_loop import _clone_message_for_send
return [_clone_message_for_send(message) for message in messages]
def finalize_turn(
agent,
*,
@@ -810,6 +819,8 @@ def finalize_turn(
and (_should_review_memory or _should_review_skills)
):
try:
# _spawn_background_review clones the snapshot structurally so
# the fork's in-place sanitizers can't reach the live transcript.
agent._spawn_background_review(
messages_snapshot=list(messages),
review_memory=_should_review_memory,
+1 -1
View File
@@ -69,7 +69,7 @@ _VERB_GROUPS: dict[str, tuple[str, str, str]] = {
"skill_view": ("read", "skill", "skills"),
"skill_manage": ("updated", "skill", "skills"),
"skills_list": ("listed skills", "time", "times"),
"todo": ("updated", "task list", "task lists"),
"todo_list": ("updated", "task list", "task lists"),
"delegate_task": ("delegated", "task", "tasks"),
"memory": ("updated", "memory", "memories"),
}
+3 -3
View File
@@ -13,8 +13,8 @@ Providers live in ``<repo>/plugins/web/<name>/`` (built-in, auto-loaded as
``plugins.enabled``).
This ABC is the SINGLE plugin-facing surface for web providers — every
provider in the tree (brave-free, ddgs, searxng, exa, parallel, keenable,
firecrawl) implements it. The legacy in-tree ``tools.web_providers.base``
provider in the tree (brave-free, ddgs, searxng, exa, parallel, tavily,
keenable, firecrawl) implements it. The legacy in-tree ``tools.web_providers.base``
ABCs were deleted in PR #25182 along with the per-vendor inline helpers
in ``tools/web_tools.py``; the response-shape contract documented below
is preserved bit-for-bit so the tool wrapper does not have to translate.
@@ -93,7 +93,7 @@ class WebSearchProvider(abc.ABC):
:meth:`search` / :meth:`extract`. The :meth:`supports_search` /
:meth:`supports_extract` capability flags let the registry route each
tool call to the right provider, and let multi-capability providers
(Firecrawl, Keenable, Exa, …) advertise multiple capabilities from a
(Firecrawl, Tavily, Exa, …) advertise multiple capabilities from a
single class.
"""
+4 -3
View File
@@ -16,7 +16,7 @@ The active provider is chosen by configuration with this precedence:
2. ``web.backend`` (shared fallback).
3. If exactly one capability-eligible provider is registered AND available,
use it.
4. Legacy preference order — ``firecrawl`` → ``parallel`` →
4. Legacy preference order — ``firecrawl`` → ``parallel`` → ``tavily`` →
``exa`` → ``searxng`` → ``brave-free`` → ``ddgs`` — filtered by
availability. Matches the historic ``tools.web_tools._get_backend()``
candidate order so installs that never set a config key keep landing
@@ -159,6 +159,7 @@ def _read_config_key(*path: str) -> Optional[str]:
_LEGACY_PREFERENCE = (
"firecrawl",
"parallel",
"tavily",
"exa",
"searxng",
"brave-free",
@@ -167,7 +168,7 @@ _LEGACY_PREFERENCE = (
# Keyless free-tier walk — strictly LAST-resort, tried only after the
# availability-filtered legacy walk finds nothing (i.e. the user has zero
# web credentials and no importable ddgs). All five vendors expose public
# web credentials and no importable ddgs). Ring vendors expose public
# anonymous free tiers (see plugins/web/keyless_mcp.py). Unpinned keyless
# traffic round-robins across the ring per request (the ring cursor lives
# in keyless_mcp; an explicit `hermes tools` pick bypasses this walk
@@ -220,7 +221,7 @@ def _resolve(configured: Optional[str], *, capability: str) -> Optional[WebSearc
supports *capability* AND ``is_available()`` reports True, return it.
3. **Legacy preference walk, filtered by availability.** Walk the
:data:`_LEGACY_PREFERENCE` order (firecrawl → parallel →
:data:`_LEGACY_PREFERENCE` order (firecrawl → parallel → tavily →
exa → searxng → brave-free → ddgs) looking for a provider whose
``supports_<capability>()`` is True AND whose ``is_available()`` is
True. Matches the historic ``tools.web_tools._get_backend()``
@@ -1,208 +0,0 @@
import fs from 'node:fs'
import path from 'node:path'
import {
buildAppEnv,
createSandbox,
launchDesktop,
type MockBackendFixture,
waitForAppReady,
writeEnvFile,
writeMockProviderConfig
} from './fixtures'
import { MOCK_REPLY, startMockServer } from '../../../tests-js/scripts/mock-server'
import { RealSessionBuilder } from './real-session-builder'
import { expect, test } from './test'
// A bot row click is "go to this bot", not "open its Bot Chat". Before the
// fix, every click resolved the canonical chat by name and opened it as a tab
// again — a Bot Chat the user had closed came back beside every newer thread
// on every bot switch, because nothing records a close (the plugin keeps no
// closed set; core's tile bucket only forgets). Now a bot whose workspace
// already holds tabs comes back to the one the user left; the forever-chat is
// re-opened only by the explicit asks (row menu "Open Bot Chat", Bots home
// "Open chat").
type Page = MockBackendFixture['page']
let fixture: MockBackendFixture | null = null
async function openBots(page: Page): Promise<void> {
const tab = page
.getByRole('button', { name: 'Bots', exact: true })
.or(page.getByRole('tab', { name: 'Bots', exact: true }))
.first()
await tab.click()
await expect(page.getByRole('button', { name: 'New bot or group chat' })).toBeVisible()
}
/** A bot's backend spawns on its first open; give the wake a real chance to
* clear before the next gesture races it. Tolerant: the mock backend can
* keep a tile's "Waking up…" notice around. */
async function settle(page: Page, timeout = 90_000): Promise<void> {
await page
.getByText(/Waking up/i)
.first()
.waitFor({ state: 'hidden', timeout })
.catch(() => undefined)
await page.waitForTimeout(500)
}
/** A first open right after a bot's backend spawned can strand on the
* profile socket (a separate, pre-existing reconnect race); a newer click
* supersedes it. Retry the gesture like a user would before giving up. */
async function openUntil(action: () => Promise<void>, expected: () => Promise<void>, attempts = 3): Promise<void> {
for (let attempt = 1; ; attempt += 1) {
await action()
try {
await expected()
return
} catch (error) {
if (attempt >= attempts) {
throw error
}
}
}
}
const SCREENSHOT_DIR = process.env.BOT_MODE_SCREENSHOT_DIR
async function snap(page: Page, name: string): Promise<void> {
if (SCREENSHOT_DIR) {
await page.screenshot({ path: `${SCREENSHOT_DIR}/${name}.png` })
}
}
/** The session tabs on the main strip (the Bots home tab may sit beside them). */
const mainTabs = (page: Page) =>
page.evaluate(() =>
[...document.querySelectorAll<HTMLElement>('[data-zone-tabstrip="grp-main"] [data-tree-tab]')]
.map(element => element.getAttribute('data-tree-tab') ?? '')
.filter(id => id.startsWith('session-tile:'))
)
/** Bots are profiles. Seeding one on disk before launch — with the mock
* provider so its own backend can answer, and a real, durable "Bot Chat"
* row (the plugin's canonical forever-chat, found by exact title) — keeps
* in-app creation and the intro turn it fires out of a scenario that is
* about the row click. With the row present, the click takes the open-as-
* tab path; without it, it would mint the chat into the workspace pane. */
async function seedBot(hermesHome: string, mockUrl: string, name: string): Promise<void> {
const dir = path.join(hermesHome, 'profiles', name)
fs.mkdirSync(dir, { recursive: true })
writeMockProviderConfig(dir, mockUrl)
writeEnvFile(dir)
const builder = await RealSessionBuilder.start(dir)
try {
await builder.createSession({ title: 'Bot Chat', turns: [`Hello ${name}`] })
} finally {
await builder.close()
}
}
test.beforeAll(async () => {
const mock = await startMockServer()
const sandbox = createSandbox('bots')
writeMockProviderConfig(sandbox.hermesHome, mock.url)
writeEnvFile(sandbox.hermesHome)
await seedBot(sandbox.hermesHome, mock.url, 'alpha')
await seedBot(sandbox.hermesHome, mock.url, 'beta')
const { app, page } = await launchDesktop(buildAppEnv(sandbox))
fixture = {
app,
page,
mock,
mockUrl: mock.url,
sandbox,
cleanup: async () => {
await app.close().catch(() => undefined)
await mock.close()
sandbox.cleanup()
}
}
await waitForAppReady(fixture, 120_000)
})
test.afterAll(async () => {
await fixture?.cleanup()
fixture = null
})
test('a bot row click returns to the open thread and does not re-open a closed Bot Chat', async () => {
test.setTimeout(300_000)
const page = fixture!.page
await openBots(page)
const alphaRow = page.getByRole('button', { name: /^alpha\b/i }).filter({ visible: true }).first()
const betaRow = page.getByRole('button', { name: /^beta\b/i }).filter({ visible: true }).first()
await expect(alphaRow).toBeVisible({ timeout: 30_000 })
await expect(betaRow).toBeVisible({ timeout: 30_000 })
const botChatTab = page.getByRole('tab', { name: /Bot Chat/ }).filter({ visible: true })
// The first click on a bot with nothing open lands on its canonical chat.
await openUntil(
() => alphaRow.click(),
() => expect(botChatTab.first()).toBeVisible({ timeout: 45_000 })
)
await settle(page, 15_000)
await snap(page, '01-first-click-opens-bot-chat')
// Close it, then start a fresh thread for Alpha (⌘/Ctrl+T — the strip's
// "+" leaves with the zone's last tab).
await botChatTab.first().hover()
await botChatTab.first().getByRole('button', { name: 'Close' }).click({ force: true })
await expect(botChatTab).toHaveCount(0)
await page.keyboard.press('Control+t')
const composer = page.locator('[data-slot="composer-root"] [contenteditable="true"]').filter({ visible: true }).first()
await expect(composer).toBeVisible({ timeout: 15_000 })
await composer.click()
await composer.fill('hello alpha thread')
await page.keyboard.press('Enter')
await expect(page.getByText('hello alpha thread').filter({ visible: true }).first()).toBeVisible({ timeout: 15_000 })
// The reply also becomes the tab's (clipped) title — match the visible copy.
await expect(page.getByText(MOCK_REPLY).filter({ visible: true }).first()).toBeVisible({ timeout: 60_000 })
await snap(page, '02-closed-bot-chat-new-thread')
const threadTabs = await mainTabs(page)
expect(threadTabs).toHaveLength(1)
const [threadTab] = threadTabs
expect(threadTab).toMatch(/^session-tile:/)
// Switch to Beta: Alpha's thread leaves the strip (scoped away, not closed).
await betaRow.click()
await expect(page.locator(`[data-zone-tabstrip="grp-main"] [data-tree-tab="${threadTab}"]`)).toHaveCount(0, {
timeout: 60_000
})
await settle(page)
// Back to Alpha: the thread is fronted, and the closed Bot Chat STAYS closed.
await alphaRow.click()
const threadTabLocator = page.locator(`[data-zone-tabstrip="grp-main"] [data-tree-tab="${threadTab}"]`)
await expect(threadTabLocator).toBeVisible({ timeout: 30_000 })
await expect(threadTabLocator).toHaveAttribute('aria-selected', 'true')
await page.waitForTimeout(3000)
await expect(botChatTab).toHaveCount(0)
expect(await mainTabs(page)).toEqual([threadTab])
await snap(page, '03-back-to-alpha-bot-chat-stays-closed')
// The explicit ask still opens the forever-chat, beside the thread.
await openUntil(
async () => {
await alphaRow.click({ button: 'right' })
await page.getByRole('menuitem', { name: 'Open Bot Chat' }).click()
},
() => expect(botChatTab.first()).toBeVisible({ timeout: 45_000 })
)
expect(await mainTabs(page)).toHaveLength(2)
expect(await mainTabs(page)).toContain(threadTab)
await snap(page, '04-explicit-open-bot-chat')
})
@@ -0,0 +1,158 @@
import fs from 'node:fs'
import path from 'node:path'
import {
buildAppEnv,
createSandbox,
launchDesktop,
type MockBackendFixture,
waitForAppReady,
writeEnvFile,
writeMockProviderConfig
} from './fixtures'
import { MOCK_REPLY, startMockServer } from '../../../tests-js/scripts/mock-server'
import { RealSessionBuilder } from './real-session-builder'
import { expect, test } from './test'
// A bot row previews the bot's canonical Bot Chat (the gateway resolves it by
// name on every roster poll). Clicking the row must land on THAT conversation.
// Before this fix a plain click fronted whatever bots-workspace tile the user
// last had open for that bot — a `+` side thread outlived every restart in
// Local Storage and won every click forever, while the row kept previewing the
// Bot Chat. The user saw the sidebar and the center describe two different
// conversations ("sessions not in sync"; support thread 1544460286084391043).
type Page = MockBackendFixture['page']
let fixture: MockBackendFixture | null = null
async function openBots(page: Page): Promise<void> {
const tab = page
.getByRole('button', { name: 'Bots', exact: true })
.or(page.getByRole('tab', { name: 'Bots', exact: true }))
.first()
await tab.click()
await expect(page.getByRole('button', { name: 'New bot or group chat' })).toBeVisible()
}
async function settle(page: Page, timeout = 90_000): Promise<void> {
await page
.getByText(/Waking up/i)
.first()
.waitFor({ state: 'hidden', timeout })
.catch(() => undefined)
await page.waitForTimeout(500)
}
async function openUntil(action: () => Promise<void>, expected: () => Promise<void>, attempts = 3): Promise<void> {
for (let attempt = 1; ; attempt += 1) {
await action()
try {
await expected()
return
} catch (error) {
if (attempt >= attempts) {
throw error
}
}
}
}
async function seedBot(hermesHome: string, mockUrl: string, name: string): Promise<void> {
const dir = path.join(hermesHome, 'profiles', name)
fs.mkdirSync(dir, { recursive: true })
writeMockProviderConfig(dir, mockUrl)
writeEnvFile(dir)
const builder = await RealSessionBuilder.start(dir)
try {
await builder.createSession({ title: 'Bot Chat', turns: [`Hello ${name}`] })
} finally {
await builder.close()
}
}
test.beforeAll(async () => {
const mock = await startMockServer()
const sandbox = createSandbox('bots-sync')
writeMockProviderConfig(sandbox.hermesHome, mock.url)
writeEnvFile(sandbox.hermesHome)
await seedBot(sandbox.hermesHome, mock.url, 'alpha')
await seedBot(sandbox.hermesHome, mock.url, 'beta')
const { app, page } = await launchDesktop(buildAppEnv(sandbox))
fixture = {
app,
page,
mock,
mockUrl: mock.url,
sandbox,
cleanup: async () => {
await app.close().catch(() => undefined)
await mock.close()
sandbox.cleanup()
}
}
await waitForAppReady(fixture, 120_000)
})
test.afterAll(async () => {
await fixture?.cleanup()
fixture = null
})
test('a bot row click lands on the Bot Chat the row previews, not a side thread', async () => {
test.setTimeout(300_000)
const page = fixture!.page
await openBots(page)
const alphaRow = page.getByRole('button', { name: /^alpha\b/i }).filter({ visible: true }).first()
const betaRow = page.getByRole('button', { name: /^beta\b/i }).filter({ visible: true }).first()
await expect(alphaRow).toBeVisible({ timeout: 30_000 })
await expect(betaRow).toBeVisible({ timeout: 30_000 })
const seededTurn = page.getByText('Hello alpha', { exact: true }).filter({ visible: true })
await openUntil(
() => alphaRow.click(),
() => expect(seededTurn.first()).toBeVisible({ timeout: 45_000 })
)
await settle(page, 15_000)
// A `+` side thread for alpha, with a real turn so it is a persisted tile.
await page.keyboard.press('Control+t')
const composer = page.locator('[data-slot="composer-root"] [contenteditable="true"]').filter({ visible: true }).first()
await expect(composer).toBeVisible({ timeout: 15_000 })
await composer.click()
await composer.fill('hello alpha thread')
await page.keyboard.press('Enter')
await expect(page.getByText(MOCK_REPLY).filter({ visible: true }).first()).toBeVisible({ timeout: 60_000 })
// Leave alpha on the side thread, go to beta, come back via the row.
await betaRow.click()
await expect(page.getByText('Hello beta', { exact: true }).filter({ visible: true }).first()).toBeVisible({
timeout: 60_000
})
await settle(page)
await alphaRow.click()
// The row previews the Bot Chat; the click must front it.
await expect(seededTurn.first()).toBeVisible({ timeout: 45_000 })
// The side thread is still open beside it (scoped to alpha), not closed.
await expect
.poll(
() =>
page.evaluate(() =>
[...document.querySelectorAll<HTMLElement>('[data-zone-tabstrip="grp-main"] [data-tree-tab]')]
.map(element => element.getAttribute('data-tree-tab') ?? '')
.filter(id => id.startsWith('session-tile:')).length
),
{ timeout: 15_000 }
)
.toBe(1)
})
@@ -0,0 +1,137 @@
import fs from 'node:fs'
import path from 'node:path'
import {
buildAppEnv,
createSandbox,
launchDesktop,
type MockBackendFixture,
waitForAppReady,
writeEnvFile,
writeMockProviderConfig
} from './fixtures'
import { MOCK_REPLY, startMockServer } from '../../../tests-js/scripts/mock-server'
import { RealSessionBuilder } from './real-session-builder'
import { expect, test } from './test'
// Every bot's canonical chat is STORED under the same title ("Bot Chat" — the
// name the gateway resolves it by), so the main tab strip captioned every open
// bot chat identically and two bots' tabs were indistinguishable (#99152). The
// tab must read the bot's display name while the stored title stays canonical.
type Page = MockBackendFixture['page']
let fixture: MockBackendFixture | null = null
async function openBots(page: Page): Promise<void> {
const tab = page
.getByRole('button', { name: 'Bots', exact: true })
.or(page.getByRole('tab', { name: 'Bots', exact: true }))
.first()
await tab.click()
await expect(page.getByRole('button', { name: 'New bot or group chat' })).toBeVisible()
}
async function openUntil(action: () => Promise<void>, expected: () => Promise<void>, attempts = 3): Promise<void> {
for (let attempt = 1; ; attempt += 1) {
await action()
try {
await expected()
return
} catch (error) {
if (attempt >= attempts) {
throw error
}
}
}
}
async function seedBot(hermesHome: string, mockUrl: string, name: string): Promise<void> {
const dir = path.join(hermesHome, 'profiles', name)
fs.mkdirSync(dir, { recursive: true })
writeMockProviderConfig(dir, mockUrl)
writeEnvFile(dir)
const builder = await RealSessionBuilder.start(dir)
try {
await builder.createSession({ title: 'Bot Chat', turns: [`Hello ${name}`] })
} finally {
await builder.close()
}
}
/** Every tab caption in the main strip (the main `workspace` tab + tiles). */
function mainStripTabTitles(page: Page): Promise<string[]> {
return page.evaluate(() =>
[...document.querySelectorAll<HTMLElement>('[data-zone-tabstrip="grp-main"] [data-tree-tab]')].map(element =>
(element.textContent ?? '').trim()
)
)
}
test.beforeAll(async () => {
const mock = await startMockServer()
const sandbox = createSandbox('bots-tabname')
writeMockProviderConfig(sandbox.hermesHome, mock.url)
writeEnvFile(sandbox.hermesHome)
await seedBot(sandbox.hermesHome, mock.url, 'alpha')
await seedBot(sandbox.hermesHome, mock.url, 'beta')
const { app, page } = await launchDesktop(buildAppEnv(sandbox))
fixture = {
app,
page,
mock,
mockUrl: mock.url,
sandbox,
cleanup: async () => {
await app.close().catch(() => undefined)
await mock.close()
sandbox.cleanup()
}
}
await waitForAppReady(fixture, 120_000)
})
test.afterAll(async () => {
await fixture?.cleanup()
fixture = null
})
test("an open Bot Chat's tab reads the bot's name, not the canonical 'Bot Chat' title", async () => {
test.setTimeout(300_000)
const page = fixture!.page
await openBots(page)
const alphaRow = page.getByRole('button', { name: /^alpha\b/i }).filter({ visible: true }).first()
await expect(alphaRow).toBeVisible({ timeout: 30_000 })
await openUntil(
() => alphaRow.click(),
() =>
expect(page.getByText('Hello alpha', { exact: true }).filter({ visible: true }).first()).toBeVisible({
timeout: 45_000
})
)
// A `+` side thread beside the Bot Chat gives the main zone a tab strip —
// the surface where every bot chat used to read "Bot Chat".
await page.keyboard.press('Control+t')
const composer = page.locator('[data-slot="composer-root"] [contenteditable="true"]').filter({ visible: true }).first()
await expect(composer).toBeVisible({ timeout: 15_000 })
await composer.click()
await composer.fill('hello alpha thread')
await page.keyboard.press('Enter')
await expect(page.getByText(MOCK_REPLY).filter({ visible: true }).first()).toBeVisible({ timeout: 60_000 })
await expect.poll(() => mainStripTabTitles(page), { timeout: 15_000 }).toHaveLength(2)
const captions = await mainStripTabTitles(page)
expect(captions.some(caption => /alpha/i.test(caption))).toBe(true)
expect(captions.some(caption => /bot chat/i.test(caption))).toBe(false)
})
@@ -0,0 +1,254 @@
import fs from 'node:fs'
import path from 'node:path'
import {
buildAppEnv,
createSandbox,
launchDesktop,
type MockBackendFixture,
waitForAppReady,
writeEnvFile,
writeMockProviderConfig
} from './fixtures'
import { startMockServer } from '../../../tests-js/scripts/mock-server'
import { RealSessionBuilder } from './real-session-builder'
import { expect, test } from './test'
// User-made sections in the Bots roster: a bot is filed by dragging it onto a
// section or through its row menu, the section is renamed through the same
// dialog shape sessions use, and deleting a section returns its bots to
// Unassigned (with an Undo toast, no confirmation). With no sections created
// the roster is the plain list it always was.
type Page = MockBackendFixture['page']
let fixture: MockBackendFixture | null = null
// BOT_SECTIONS_SCREENSHOT_DIR=<dir> saves full-window captures at the key
// states — handy for design review; never part of the assertions.
async function capture(page: Page, name: string): Promise<void> {
const dir = process.env.BOT_SECTIONS_SCREENSHOT_DIR
if (!dir) {
return
}
fs.mkdirSync(dir, { recursive: true })
await page.screenshot({ path: path.join(dir, `${name}.png`) })
}
async function seedBot(hermesHome: string, mockUrl: string, name: string): Promise<void> {
const dir = path.join(hermesHome, 'profiles', name)
fs.mkdirSync(dir, { recursive: true })
writeMockProviderConfig(dir, mockUrl)
writeEnvFile(dir)
const builder = await RealSessionBuilder.start(dir)
try {
await builder.createSession({ title: 'Bot Chat', turns: [`Hello ${name}`] })
} finally {
await builder.close()
}
}
const roster = (page: Page) => page.locator('[data-slot="bots-roster"]')
const botRow = (page: Page, name: string) => roster(page).locator(`[data-roster-key="local::${name}"]`)
/** A section's label span — the one node whose text is exactly the name. */
const sectionLabel = (page: Page, name: string) =>
page.locator('span.truncate', { hasText: new RegExp(`^${name}$`, 'i') })
/** The heading's fold button (label + count) — the ⋯ menu trigger is a sibling with no text. */
const sectionHeading = (page: Page, name: string) =>
roster(page).locator('[data-slot="bots-section"] button[aria-expanded]').filter({ has: sectionLabel(page, name) })
const sectionBlock = (page: Page, name: string) =>
roster(page).locator('[data-slot="bots-section"]').filter({ has: sectionLabel(page, name) })
/** Section name → roster keys of the rows under it (the plain list has no sections). */
async function layout(page: Page): Promise<Array<[string, string[]]>> {
return roster(page).locator('[data-slot="bots-section"]').evaluateAll(blocks =>
blocks.map(block => [
block.querySelector('button[aria-expanded] span.truncate')?.textContent?.trim() ?? '',
[...block.querySelectorAll<HTMLElement>('[data-roster-key]')].map(row => row.dataset.rosterKey ?? '')
])
)
}
test.beforeAll(async () => {
const mock = await startMockServer()
const sandbox = createSandbox('bots-sections')
writeMockProviderConfig(sandbox.hermesHome, mock.url)
writeEnvFile(sandbox.hermesHome)
for (const name of ['alpha', 'beta', 'gamma']) {
await seedBot(sandbox.hermesHome, mock.url, name)
}
const { app, page } = await launchDesktop(buildAppEnv(sandbox))
fixture = {
app,
page,
mock,
mockUrl: mock.url,
sandbox,
cleanup: async () => {
await app.close().catch(() => undefined)
await mock.close()
sandbox.cleanup()
}
}
await waitForAppReady(fixture, 120_000)
})
test.afterAll(async () => {
await fixture?.cleanup()
fixture = null
})
test('file bots into user sections by menu and drag; rename; delete returns them to Unassigned', async () => {
test.setTimeout(300_000)
const page = fixture!.page
const tab = page
.getByRole('button', { name: 'Bots', exact: true })
.or(page.getByRole('tab', { name: 'Bots', exact: true }))
.first()
await tab.click()
await expect(page.getByRole('button', { name: 'New bot or group chat' })).toBeVisible()
await expect(botRow(page, 'alpha')).toBeVisible({ timeout: 30_000 })
await expect(botRow(page, 'beta')).toBeVisible({ timeout: 30_000 })
// No sections yet: the plain list, no section chrome at all.
await expect(roster(page).locator('[data-slot="bots-section"]')).toHaveCount(0)
await capture(page, '1-plain-roster')
// Right-click alpha → Move to section → New section… → name it → alpha is filed.
await botRow(page, 'alpha').click({ button: 'right' })
await page.getByRole('menuitem', { name: 'Move to section' }).hover()
await expect(page.getByRole('menuitem', { name: 'New section…' })).toBeVisible()
await capture(page, '2-row-menu-move-to-section')
await page.getByRole('menuitem', { name: 'New section…' }).click()
const nameField = page.getByRole('textbox', { name: 'Section name' })
await expect(nameField).toBeVisible()
await nameField.fill('Clients')
await capture(page, '3-new-section-dialog')
await page.getByRole('button', { name: 'Create' }).click()
await expect(sectionHeading(page, 'Clients')).toBeVisible()
await expect(sectionBlock(page, 'Clients').locator('[data-roster-key="local::alpha"]')).toBeVisible()
// The remainder is Unassigned, drawn last.
await expect
.poll(async () => (await layout(page)).map(([name, keys]) => [name, keys.length]))
.toEqual([
['Clients', 1],
['Unassigned', 3]
])
await capture(page, '4-alpha-filed')
// Drag beta over the Clients block: the target highlights while over it.
// Escape cancels — nothing moves, nothing stays highlighted or faded.
const target = sectionBlock(page, 'Clients')
const from = (await botRow(page, 'beta').boundingBox())!
const to = (await sectionHeading(page, 'Clients').boundingBox())!
const dragBetaOverClients = async () => {
await page.mouse.move(from.x + from.width / 2, from.y + from.height / 2)
await page.mouse.down()
await page.mouse.move(from.x + from.width / 2, from.y + from.height / 2 - 10, { steps: 4 })
await page.mouse.move(to.x + to.width / 2, to.y + to.height / 2, { steps: 12 })
await expect(target).toHaveAttribute('data-drop-over', 'true')
}
await dragBetaOverClients()
await page.keyboard.press('Escape')
await page.mouse.up()
await expect(target).not.toHaveAttribute('data-drop-over', 'true')
await expect(botRow(page, 'beta')).toHaveCSS('opacity', '1')
expect((await layout(page)).map(([name, keys]) => [name, keys.length])).toEqual([
['Clients', 1],
['Unassigned', 3]
])
// Drop it for real: the bot is filed.
await dragBetaOverClients()
await capture(page, '5-drag-over-clients')
await page.mouse.up()
await expect(target.locator('[data-roster-key="local::beta"]')).toBeVisible()
await expect(target).not.toHaveAttribute('data-drop-over', 'true')
// The moved row remounts under its new section; it must not stay faded.
await expect(botRow(page, 'beta')).toHaveCSS('opacity', '1')
await expect
.poll(async () => (await layout(page)).map(([name, keys]) => [name, keys.length]))
.toEqual([
['Clients', 2],
['Unassigned', 2]
])
await capture(page, '6-beta-dropped')
// Rename through the heading's context menu — the same Dialog + Input
// + Save shape as a session rename.
await sectionHeading(page, 'Clients').click({ button: 'right' })
await page.getByRole('menuitem', { name: 'Rename…' }).click()
await expect(nameField).toHaveValue('Clients')
await nameField.fill('Customers')
await page.getByRole('button', { name: 'Save' }).click()
await expect(sectionHeading(page, 'Customers')).toBeVisible()
await expect(sectionHeading(page, 'Clients')).toHaveCount(0)
await capture(page, '7-renamed')
// A second, empty section from the + menu shows its drop hint; collapsing
// a section folds its rows like the gateway headings do.
await page.getByRole('button', { name: 'New bot or group chat' }).click()
await page.getByRole('menuitem', { name: 'New section' }).click()
await nameField.fill('Team')
await page.getByRole('button', { name: 'Create' }).click()
await expect(sectionBlock(page, 'Team').getByText('Drag bots here')).toBeVisible()
await sectionHeading(page, 'Customers').click()
await expect(sectionBlock(page, 'Customers').locator('[data-roster-key]')).toHaveCount(0)
await capture(page, '8-empty-section-and-collapsed')
await sectionHeading(page, 'Customers').click()
await expect(sectionBlock(page, 'Customers').locator('[data-roster-key]')).toHaveCount(2)
// Delete Customers: no confirmation, its two bots return to Unassigned,
// and the toast offers Undo.
await sectionHeading(page, 'Customers').click({ button: 'right' })
await page.getByRole('menuitem', { name: 'Delete' }).click()
await expect(sectionHeading(page, 'Customers')).toHaveCount(0)
const toast = page.getByRole('status').filter({ hasText: 'Deleted “Customers”' })
await expect(toast).toBeVisible()
await expect
.poll(async () => (await layout(page)).map(([name, keys]) => [name, keys.length]))
.toEqual([
['Team', 0],
['Unassigned', 4]
])
await capture(page, '9-deleted-with-undo-toast')
await toast.getByRole('button', { name: 'Undo' }).click()
await expect(sectionHeading(page, 'Customers')).toBeVisible()
await expect
.poll(async () => (await layout(page)).map(([name, keys]) => [name, keys.length]))
.toEqual([
['Customers', 2],
['Team', 0],
['Unassigned', 2]
])
// Membership rides the bot's profile ui_meta, so it follows profile sync.
const alphaProfile = path.join(fixture!.sandbox.hermesHome, 'profiles', 'alpha', 'profile.yaml')
await expect.poll(() => (fs.existsSync(alphaProfile) ? fs.readFileSync(alphaProfile, 'utf8') : '')).toMatch(/sectionId:\s*sec-/)
// Delete both sections: the roster is the plain list again.
for (const name of ['Customers', 'Team']) {
await sectionHeading(page, name).click({ button: 'right' })
await page.getByRole('menuitem', { name: 'Delete' }).click()
}
await expect(roster(page).locator('[data-slot="bots-section"]')).toHaveCount(0)
await expect(botRow(page, 'alpha')).toBeVisible()
})
+6 -5
View File
@@ -106,7 +106,10 @@ test.describe('chat interaction with mock backend', () => {
await composer.click()
await composer.type('please answer tersely')
await expect(primary).toHaveAttribute('aria-label', /Steer/)
// Since "running is not busy" (3bc52fb9df) the primary keeps the Send
// affordance mid-turn — steer is routed through the submit engine, not a
// separate labeled button. Queue remains the explicit secondary action.
await expect(primary).toHaveAttribute('aria-label', 'Send')
await expect(dictation).toBeVisible()
await expect(speakReplies).toBeVisible()
await expect(queue).toBeVisible()
@@ -119,11 +122,9 @@ test.describe('chat interaction with mock backend', () => {
)
expect(controlLabels.indexOf('Voice dictation')).toBeLessThan(speakRepliesIndex)
expect(speakRepliesIndex).toBeLessThan(controlLabels.indexOf('Queue message'))
expect(controlLabels.indexOf('Queue message')).toBeLessThan(
controlLabels.findIndex(label => label?.startsWith('Steer'))
)
expect(controlLabels.indexOf('Queue message')).toBeLessThan(controlLabels.indexOf('Send'))
await page.screenshot({ path: testInfo.outputPath('busy-composer-steer.png') })
await expect(primary.locator('svg.tabler-icon-steering-wheel')).toBeVisible()
await expect(primary.locator('.codicon-arrow-up')).toBeVisible()
await queue.click()
await expect(primary).toHaveAttribute('aria-label', 'Stop')
@@ -45,7 +45,9 @@ async function steer(page: Page, text: string): Promise<void> {
await composer.waitFor({ state: 'visible', timeout: 15_000 })
await composer.click()
await composer.type(text, { delay: 5 })
await expect(primary).toHaveAttribute('aria-label', /Steer/)
// Since "running is not busy" (3bc52fb9df) the primary keeps the Send label
// mid-turn; the submit engine still routes a text payload to steer.
await expect(primary).toHaveAttribute('aria-label', 'Send')
await primary.click()
}
@@ -209,17 +211,36 @@ test.describe('correction session switch', () => {
// Reproduce the observed race: switch to another persisted session while
// the foreground tool is live, then return before its redirect settles.
await openSidebarSession(page, MOCK_REPLY, OTHER_SESSION_PROMPT)
// Sidebar rows title by the session's first user prompt (auto-title is
// disabled in the e2e fixture config).
await openSidebarSession(page, OTHER_SESSION_PROMPT, OTHER_SESSION_PROMPT)
await reopenOriginalSession(page)
await page.waitForTimeout(500)
// The warm resume first paints the persisted history and then reconciles
// the live turn (including a steer whose persistence may lag on a loaded
// runner) back in. Poll to the converged order instead of sampling one
// arbitrary mid-reconcile frame; the duplicate checks then pin the
// regression (the prompt/correction must appear exactly once).
await expect
.poll(async () => relevantOrder(await transcriptTextOrder(page)), {
message: 'correction should stay in place after the warm resume',
timeout: 30_000,
})
.toEqual(orderBeforeSwitch)
await page.screenshot({ path: testInfo.outputPath('correction-after-warm-resume.png') })
expect(relevantOrder(await transcriptTextOrder(page))).toEqual(orderBeforeSwitch)
expect(await textNodeOccurrences(page, ORIGINAL_PROMPT)).toBe(1)
expect(await textNodeOccurrences(page, CORRECTION)).toBe(1)
await waitForTranscriptText(page, CORRECTED_REPLY)
expect(steerTurnOrder(await transcriptMessageOrder(page))).toEqual([ORIGINAL_PROMPT, CORRECTION, CORRECTED_REPLY])
// The post-turn stored-history reconcile can momentarily repaint from a
// snapshot in which the steer's user row hasn't been folded back in yet —
// poll to the converged order instead of sampling one frame.
await expect
.poll(async () => steerTurnOrder(await transcriptMessageOrder(page)), {
message: 'steered turn should settle as prompt → correction → corrected reply',
timeout: 30_000,
})
.toEqual([ORIGINAL_PROMPT, CORRECTION, CORRECTED_REPLY])
})
test('keeps an inference-time correction visible through a warm session switch', async ({}, testInfo: TestInfo) => {
@@ -236,7 +257,7 @@ test.describe('correction session switch', () => {
await send(page, INFERENCE_CORRECTION)
await waitForTranscriptText(page, INFERENCE_CORRECTION)
await openSidebarSession(page, MOCK_REPLY, OTHER_SESSION_PROMPT)
await openSidebarSession(page, OTHER_SESSION_PROMPT, OTHER_SESSION_PROMPT)
await reopenInferenceSession(page)
expect(await textNodeOccurrences(page, INFERENCE_PROMPT)).toBe(1)
+24 -1
View File
@@ -170,6 +170,29 @@ export function writeMockProviderConfig(
? `\ndisplay:\n${extraDisplayConfig}\n`
: ''
// Title generation rides the MAIN model since 87af576e60 (#83636), so every
// completed turn fires an extra background /v1/chat/completions at the mock.
// That request contains the whole conversation — trigger keywords included —
// which advances the mock's scripted-turn indices and trips hold-for-prompt
// matchers from a request no spec ever sent. Disable it by default (no e2e
// spec asserts on session titles); a test that passes its own `auxiliary:`
// section via extraConfig owns the whole section instead.
const autoTitleDefault = extraConfig?.includes('auxiliary:')
? ''
: 'auxiliary:\n title_generation:\n enabled: false\n'
// The scripted turns run REAL terminal commands, and anything the guard
// classifies as dangerous (e.g. the sidebar sentinel-wait loop) parks the
// turn behind a Run/Reject approval card. The default 'smart' mode then
// fires an aux LLM approval call at the SAME mock provider — consuming a
// scripted-turn index and never resolving — so the turn stalls until the
// spec times out (the CI failure mode for the sidebar-dot family). No e2e
// spec asserts on the approval flow, so run gate-free by default; a test
// that passes its own `approvals:` section via extraConfig owns it.
const approvalsDefault = extraConfig?.includes('approvals:')
? ''
: 'approvals:\n mode: "off"\n'
const config = `# Auto-generated by E2E test fixtures
model:
default: mock-model
@@ -183,7 +206,7 @@ ${modelContextLength ? ` context_length: ${modelContextLength}\n` : ''}provider
models:
mock-model: {}
context_length: 4096
${displaySection}${extraConfig ? `\n${extraConfig.trim()}\n` : ''}`
${autoTitleDefault}${approvalsDefault}${displaySection}${extraConfig ? `\n${extraConfig.trim()}\n` : ''}`
fs.writeFileSync(configPath, config, 'utf8')
}
+39 -8
View File
@@ -24,20 +24,46 @@ import { expect, type Page, test } from '@playwright/test'
import { type MockBackendFixture, setupMockBackend, waitForAppReady } from './fixtures'
const STRIP = '.glyph-spinner__strip'
/* Scope to a spinner that is actually RUNNING. Turns from earlier tests in
* this file leave parked spinners mounted (kept-alive panes, swap overlays
* hold them with data-paused='true'), and document.querySelector returns the
* FIRST strip in the DOM — a stale parked one once two turns have run. */
const STRIP = '.glyph-spinner:not([data-paused="true"]) .glyph-spinner__strip'
/** Prompt the mock server holds open so the spinner runs for the whole file. */
const SPINNER_PROMPT = 'E2E_GLYPH_SPINNER_HOLD'
/**
* Send a message so a turn is in flight — the composer status stack mounts a
* GlyphSpinner while the agent is working. Resolves once a frame strip is in
* the DOM.
* Get a RUNNING frame strip into the DOM deterministically.
*
* A turn is sent so the app is genuinely busy (the mock server holds the
* stream open), but which surface mounts a spinner mid-turn is app policy
* that has changed before and will again — the transcript, status stack and
* swap overlay all park/unmount theirs at different moments, which made this
* spec racy. The contract under test is the STYLESHEET (steps() animation,
* layer promotion, the data-paused and global pause gates), and that CSS is
* driven entirely by the `data-paused` attribute — the same attribute the
* parked assertions below already toggle. So: wait for any mounted spinner
* (the ChatSwapOverlay keeps one mounted, parked, after boot), then unpark it
* and assert against the running animation.
*/
async function mountSpinner(page: Page): Promise<void> {
if (await page.locator(STRIP).count()) {
return
}
const composer = page.locator('[contenteditable="true"]').first()
await composer.waitFor({ state: 'visible', timeout: 10_000 })
await composer.click()
await composer.type('hello from the glyph spinner spec', { delay: 10 })
await composer.type(SPINNER_PROMPT, { delay: 10 })
await page.keyboard.press('Enter')
await page.waitForSelector('.glyph-spinner__strip', { state: 'attached', timeout: 20_000 })
await page.evaluate(() => {
for (const el of document.querySelectorAll('.glyph-spinner[data-paused]')) {
el.removeAttribute('data-paused')
}
})
await page.waitForSelector(STRIP, { state: 'attached', timeout: 20_000 })
}
@@ -45,11 +71,14 @@ test.describe('GlyphSpinner (compositor animation)', () => {
let fixture: MockBackendFixture
test.beforeAll(async () => {
fixture = await setupMockBackend()
fixture = await setupMockBackend({
mockServer: { holdFirstStreamForPrompt: SPINNER_PROMPT },
})
await waitForAppReady(fixture)
})
test.afterAll(async () => {
fixture?.mock.releaseHeldStream()
await fixture?.cleanup()
})
@@ -93,8 +122,10 @@ test.describe('GlyphSpinner (compositor animation)', () => {
// multiple of the frame count — not the single-frame interval.
expect(observed.durationMs).toBeGreaterThan(0)
// Length-typed travel, never a percentage: `translateY(-100%)` would keep
// the animation off the compositor.
expect(observed.travel).toContain('calc(')
// the animation off the compositor. Chromium has serialized the resolved
// keyframe both as the authored `calc(...)` and as an absolute `...px`
// length depending on version — accept any length, reject percentages.
expect(observed.travel).toMatch(/calc\(|px\)/)
expect(observed.travel).not.toContain('%')
})
@@ -32,7 +32,7 @@ test.afterAll(async () => {
})
test('local bot replaces an open group main workspace', async () => {
test.setTimeout(180_000)
test.setTimeout(240_000)
const page = fixture!.page
await openBots(page)
@@ -60,11 +60,21 @@ test('local bot replaces an open group main workspace', async () => {
const programmer = page.getByRole('button', { name: /^Programmer\b/ }).filter({ visible: true }).first()
await programmer.click()
const botChatTab = page.getByRole('tab', { name: /Bot Chat Close/ }).filter({ visible: true })
await expect(botChatTab).toBeVisible({ timeout: 30_000 })
await expect(botChatTab).toHaveAttribute('aria-selected', 'true')
// The bot's canonical chat opens INTO the main workspace pane (post
// design-system rework); as the lone pane in the zone it renders chromeless
// — no "Bot Chat" tab exists until a second pane joins the strip. The
// handoff is observed by the group surfaces leaving and the bot's chat
// (here a fresh one: its empty-state splash asks for a first message)
// taking the main workspace. The first open also spawns the bot's own
// backend, so give the "Loading session" phase a real chance to clear.
await expect(page.getByText('Say something to get started.').filter({ visible: true })).toBeVisible({
timeout: 120_000
})
await expect(groupTab).toHaveCount(0)
await expect(groupComposer).toHaveCount(0)
await expect(page.getByText(/Waking up Programmer/i)).toHaveCount(0)
// No "Waking up…" assertion: the mock backend can keep a bot's wake notice
// around indefinitely (see bot-mode-row-click-mirrors-registry's settle()),
// so its presence no longer distinguishes a stranded handoff. The splash
// and composer above are the proof the bot's chat took the workspace.
await expect(page.locator('[data-slot="composer-root"] [contenteditable="true"]').filter({ visible: true }).first()).toBeVisible()
})
@@ -9,13 +9,11 @@
import * as fs from 'node:fs'
import * as path from 'node:path'
import { expect, test } from './test'
import {
type MockBackendFixture,
buildAppEnv,
createSandbox,
launchDesktop,
type MockBackendFixture,
waitForAppReady,
writeEnvFile,
writeMockProviderConfig,
@@ -27,6 +25,7 @@ import {
VERIFICATION_STOP_TRIGGER,
} from '../../../tests-js/scripts/mock-server'
import { RealSessionBuilder } from './real-session-builder'
import { expect, test } from './test'
const SESSION_TITLE = 'E2E Hidden History Messages'
const VISIBLE_USER_TEXT = 'E2E_VISIBLE_USER_HISTORY'
@@ -44,6 +43,7 @@ async function setupSeededMockBackend(): Promise<MockBackendFixture> {
)
writeEnvFile(sandbox.hermesHome)
const builder = await RealSessionBuilder.start(sandbox.hermesHome)
try {
await builder.createSession({
title: SESSION_TITLE,
@@ -83,6 +83,7 @@ test('resume hides real context-compaction handoffs', async ({}, testInfo) => {
.locator('[data-slot="sidebar"] button')
.filter({ hasText: SESSION_TITLE })
.first()
await sessionRow.click()
const transcript = page.locator('[data-slot="aui_thread-viewport"]')
@@ -110,8 +111,20 @@ test('live verify-on-stop continuations stay out of the transcript', async ({},
const mock = await startMockServer({ verificationWritePath: changedFile })
writeMockProviderConfig(sandbox.hermesHome, mock.url)
fs.appendFileSync(path.join(sandbox.hermesHome, 'config.yaml'), '\nagent:\n verify_on_stop: true\n', 'utf8')
// Auto session titling (feat f726090d48) fires an auxiliary title_generation
// LLM call whose user snippet CONTAINS the trigger keyword, so the mock's
// isVerificationStopTrigger matches it and the title call steals a scripted
// verify-on-stop turn (the transcript then ends on 'The code edit is
// complete.' instead of the exhausted-verifier final). Disable the
// model-backed title upgrade so script indices track real chat turns.
fs.appendFileSync(
path.join(sandbox.hermesHome, 'config.yaml'),
'\nauxiliary:\n title_generation:\n enabled: false\n',
'utf8',
)
writeEnvFile(sandbox.hermesHome)
const { app, page } = await launchDesktop(buildAppEnv(sandbox))
const fixture: MockBackendFixture = {
app,
page,
@@ -27,8 +27,8 @@ import { type MockServer, startMockServer } from '../../../tests-js/scripts/mock
import { RealSessionBuilder } from './real-session-builder'
import { type ElectronApplication, expect, type Page, test } from './test'
// A seeded session has no generated title, so every label falls back to the
// session preview — the first 60 characters of the first user message.
// The builder-provided title now labels the sidebar row directly (seeded
// sessions no longer fall back to the first-user-message preview).
const SESSION_TITLE = 'E2E attached image session'
const CAPTION = 'E2E attached image must survive a relaunch'
const IMAGE_DIR = 'Application Support/e2e shots'
@@ -90,7 +90,7 @@ async function setupSeededDesktop(): Promise<SeededFixture> {
}
function sessionRow(page: Page) {
return page.locator('[data-slot="sidebar"] button').filter({ hasText: CAPTION }).first()
return page.locator('[data-slot="sidebar"] button').filter({ hasText: SESSION_TITLE }).first()
}
// Inactive tabs stay mounted under a data-pane-hidden ancestor. Match the
@@ -172,13 +172,15 @@ test.describe('attached image resume', () => {
fixture = await setupSeededDesktop()
await waitForAppReady(fixture, 120_000)
// The sidebar labels a session by its preview, so the caption has to lead
// the persisted turn — a leading directive reads as a truncated file path.
// The sidebar labels a seeded session by its title. Whatever the label
// source, an attachment directive must never leak into it as a file path.
const row = sessionRow(fixture.page)
await row.waitFor({ state: 'visible', timeout: 60_000 })
const label = (await row.textContent())?.trim() ?? ''
expect(label.startsWith(CAPTION), `sidebar label should open with the caption: ${label}`).toBe(true)
expect(label.startsWith(SESSION_TITLE), `sidebar label should open with the title: ${label}`).toBe(true)
expect(label, `sidebar label should not leak the image path: ${label}`).not.toContain(IMAGE_NAME)
expect(label, `sidebar label should not render the directive: ${label}`).not.toContain('@image:')
await openSeededSession(fixture.page)
await assertRendersThumbnail(fixture.page, 'first open')
+80 -20
View File
@@ -20,16 +20,24 @@
*
* display.interim_assistant_messages: true (default)
* → ALL interim texts AND the final text must be visible in the
* transcript.
* settled transcript.
*
* display.interim_assistant_messages: false
* → only the final text is visible (no message.interim events emitted,
* so all streamed interim text is replaced at message.complete).
* → no message.interim events are emitted, so no sealed interim bubbles
* are created while streaming. Since the post-turn stored-history
* reconcile (sessions.changed → reconcileActiveTranscript, commit
* 1a2b0ca8cb) converges the visible transcript to the persisted
* transcript — which has ALWAYS contained the mid-turn commentary as
* real assistant rows (that is what a resume shows, flag or no flag) —
* the settled DOM shows the whole turn as ONE assistant message
* containing commentary + final. The flag governs live sealing only.
* The test pins that converged single-message shape: every text
* appears exactly once, inside a single assistant message root.
*
* Prerequisite: `npm run build` must have been run so dist/ exists.
*/
import { expect, test, type Page } from '@playwright/test'
import { expect, type Page, test } from '@playwright/test'
import {
type MockBackendFixture,
@@ -40,6 +48,17 @@ import { INTERIM_TEXTS, restartMockServer } from '../../../tests-js/scripts/mock
// ─── Helpers ──────────────────────────────────────────────────────────
/**
* Auto session titling (feat f726090d48, 2026-08-08) issues an auxiliary
* `title_generation` LLM call against the SAME provider as the chat turn.
* The mock server counts every completion request as a script turn, so the
* title call races the chat turn and steals a scripted interim turn (the
* stolen turn's text then never streams to the transcript). Disable the
* model-backed title upgrade — the instant derived title needs no LLM call —
* so the mock's script indices line up with real chat turns again.
*/
const DISABLE_AUTO_TITLE = 'auxiliary:\n title_generation:\n enabled: false'
/** Unique trigger keyword the mock server detects to switch to the script. */
const TRIGGER = 'E2E_INTERIM_TRIGGER'
@@ -72,7 +91,7 @@ async function sendInterimMessage(page: Page): Promise<void> {
)
// Give the renderer a moment to settle any final state updates
// (hydration, session refresh) before asserting.
// (hydration, stored-history reconcile, session refresh) before asserting.
await page.waitForTimeout(2000)
}
@@ -90,11 +109,13 @@ async function countTranscriptMessagesContaining(page: Page, text: string): Prom
return page.evaluate(
(search) => {
const viewport = document.querySelector('[data-slot="aui_thread-viewport"]')
if (!viewport) {
return 0
}
let count = 0
const walker = document.createTreeWalker(
viewport,
NodeFilter.SHOW_ELEMENT,
@@ -102,29 +123,46 @@ async function countTranscriptMessagesContaining(page: Page, text: string): Prom
acceptNode: (node) => {
const el = node as HTMLElement
const directText = el.textContent ?? ''
if (!directText.includes(search)) {
return NodeFilter.FILTER_SKIP
}
// Only count leaf-ish elements to avoid double-counting.
const hasChildWithText = Array.from(el.children).some(
(child) => (child.textContent ?? '').includes(search),
)
if (hasChildWithText) {
return NodeFilter.FILTER_SKIP
}
return NodeFilter.FILTER_ACCEPT
},
},
)
while (walker.nextNode()) {
count++
}
return count
},
text,
)
}
/** Count assistant message roots in the settled transcript. */
async function countAssistantMessageRoots(page: Page): Promise<number> {
return page.evaluate(() => {
const viewport = document.querySelector('[data-slot="aui_thread-viewport"]')
return viewport
? viewport.querySelectorAll('[data-slot="aui_assistant-message-root"]').length
: 0
})
}
// ─── Flag ON: interim_assistant_messages = true (default) ─────────────
test.describe('interim assistant messages — flag ON (default)', () => {
@@ -134,7 +172,7 @@ test.describe('interim assistant messages — flag ON (default)', () => {
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend()
fixture = await setupMockBackend({ extraConfig: DISABLE_AUTO_TITLE })
await waitForAppReady(fixture, 120_000)
})
@@ -147,8 +185,10 @@ test.describe('interim assistant messages — flag ON (default)', () => {
await sendInterimMessage(page)
// Every interim text (turns with visible text + tool calls) must be
// present in the transcript as its own sealed message — NOT wiped by
// message.complete.
// present in the settled transcript — NOT wiped by message.complete.
// (Live, each seals as its own bubble; the post-turn stored-history
// reconcile then converges the turn into one assistant message that
// still carries all of them.)
for (const interimText of INTERIM_TEXTS.interims) {
await expect
.poll(
@@ -165,6 +205,13 @@ test.describe('interim assistant messages — flag ON (default)', () => {
{ timeout: 15_000, message: 'final text should be visible' },
)
.toBeGreaterThanOrEqual(1)
// No duplicates: the reconcile must CONVERGE (replace the sealed live
// bubbles), never render a stored copy alongside a live one.
for (const text of [...INTERIM_TEXTS.interims, INTERIM_TEXTS.finalText]) {
const count = await countTranscriptMessagesContaining(page, text)
expect(count, `"${text}" must not be duplicated after reconcile`).toBe(1)
}
})
})
@@ -179,6 +226,7 @@ test.describe('interim assistant messages — flag OFF', () => {
restartMockServer()
fixture = await setupMockBackend({
extraDisplayConfig: ' interim_assistant_messages: false',
extraConfig: DISABLE_AUTO_TITLE,
})
await waitForAppReady(fixture, 120_000)
})
@@ -187,7 +235,7 @@ test.describe('interim assistant messages — flag OFF', () => {
await fixture?.cleanup()
})
test('only the final response is visible; all interim texts are wiped', async () => {
test('settled transcript converges to stored history as a single turn message', async () => {
const page = fixture.page
await sendInterimMessage(page)
@@ -199,17 +247,29 @@ test.describe('interim assistant messages — flag OFF', () => {
)
.toBeGreaterThanOrEqual(1)
// NONE of the interim texts should be visible — with the flag off,
// the tui_gateway never installs interim_assistant_callback, so no
// message.interim events are emitted. All streamed interim text is
// accumulated into the streaming bubble and replaced by
// message.complete.
for (const interimText of INTERIM_TEXTS.interims) {
const count = await countTranscriptMessagesContaining(page, interimText)
expect(
count,
`interim text "${interimText}" should NOT be visible when flag is off`,
).toBe(0)
// With the flag off, the tui_gateway never installs
// interim_assistant_callback, so no message.interim events fire and no
// sealed interim bubbles are created while streaming. After
// message.complete, the stored-history reconcile (sessions.changed →
// reconcileActiveTranscript) converges the view to the persisted
// transcript, which contains the mid-turn commentary as real assistant
// rows — exactly what a resume of this session would show. Pin that
// converged shape: ONE assistant message root for the whole turn…
await expect
.poll(
() => countAssistantMessageRoots(page),
{ timeout: 15_000, message: 'the settled turn should render as one assistant message' },
)
.toBe(1)
// …containing every commentary text and the final text exactly once.
for (const text of [...INTERIM_TEXTS.interims, INTERIM_TEXTS.finalText]) {
await expect
.poll(
() => countTranscriptMessagesContaining(page, text),
{ timeout: 15_000, message: `"${text}" should appear exactly once in the converged turn` },
)
.toBe(1)
}
})
})
@@ -57,6 +57,23 @@ test.describe('session compression', () => {
await send(page, 'E2E_COMPRESSION_THIRD')
await expect.poll(() => receivedUserTexts().filter(text => text === 'E2E_COMPRESSION_THIRD').length).toBe(1)
// The mock receiving the third prompt does not mean the TURN is over —
// /compress on a busy session errors with "session busy — /interrupt the
// current turn before /compress". Wait for the third reply to render and
// for the composer to leave its busy state (no Stop affordance) first.
await page.waitForFunction(
expected =>
((document.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').split(expected).length - 1) >= 3,
reply,
{ timeout: 90_000 }
)
await expect
.poll(
() => page.locator('[data-slot="composer-root"] button[aria-label="Stop"]').count(),
{ timeout: 30_000, message: 'turn should settle before /compress' }
)
.toBe(0)
// This test covers compression and continuation, not slash completion.
// Insert the complete command atomically and click Send so an async
// completion response cannot consume Enter as a picker acceptance.
@@ -89,6 +106,8 @@ test.describe('session compression in progress', () => {
protect_first_n: 0
protect_last_n: 1
auxiliary:
title_generation:
enabled: false
compression:
provider: custom
model: mock-model`,
@@ -124,7 +143,11 @@ auxiliary:
await expect(page.getByRole('status', { name: 'Summarizing thread' }).last()).toBeVisible()
const primary = page.locator('[data-slot="composer-root"] button[type="submit"]')
await expect(primary).toHaveAttribute('aria-label', 'Queue message')
// Since "running is not busy" (3bc52fb9df) an empty composer mid-turn
// shows Stop — the Queue affordance appears once a payload is typed, and
// the Enter path below still queues instead of steering while compaction
// holds the turn.
await expect(primary).toHaveAttribute('aria-label', 'Stop')
await send(page, queued)
await expect(page.getByText('1 Queued')).toBeVisible()
+76 -33
View File
@@ -30,6 +30,18 @@ const SESSION_RUNNING_DOT_LABEL = 'Session running'
/** Finished-unread dot aria-label. */
const UNREAD_DOT_LABEL = 'Finished — unread'
/**
* The auto-title auxiliary call hits the SAME mock provider as the chat turn,
* and its request carries the user's message — trigger keyword included. The
* mock's trigger matching is text-based, so the title call consumes a script
* index: the real chat turn then gets turn 2 (final answer, NO tool calls),
* the background process is never spawned, and the bg dot never appears.
* Whether that happens depends on which request lands first — the CI flake
* these specs had. Disable auto-title so script indices line up with real
* chat turns (same fix as interim-messages.spec.ts).
*/
const DISABLE_AUTO_TITLE = 'auxiliary:\n title_generation:\n enabled: false'
/** Send a message and wait for the final response to appear. */
async function sendMessageAndWait(
page: Page,
@@ -67,7 +79,7 @@ test.describe('sidebar states — background process and subagent', () => {
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend()
fixture = await setupMockBackend({ extraConfig: DISABLE_AUTO_TITLE })
await waitForAppReady(fixture, 120_000)
})
@@ -120,22 +132,32 @@ test.describe('sidebar states — subagent and background dot coexist', () => {
test.describe.configure({ mode: 'serial' })
let fixture: MockBackendFixture
// Hold the background process open until the test releases it. Without the
// sentinel the process is a bare `sleep 5` racing the agent turn (two model
// trips + a real subagent spawn): on a loaded runner the turn outlives the
// sleep, the process is reaped mid-turn, and the dot never appears at all —
// the CI flake this spec had.
const bgRelease = createBackgroundReleaseHandle()
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend()
fixture = await setupMockBackend({
extraConfig: DISABLE_AUTO_TITLE,
mockServer: { backgroundReleasePath: bgRelease.path },
})
await waitForAppReady(fixture, 120_000)
})
test.afterAll(async () => {
bgRelease.release()
await fixture?.cleanup()
bgRelease.cleanup()
})
test('background dot visible while subagent runs', async () => {
const page = fixture.page
// Start the turn but DON'T wait for the final answer yet — we want
// to assert the background dot is visible WHILE the subagent runs.
// Start the turn — a held background process plus a real subagent.
const composer = page.locator('[contenteditable="true"]').first()
await composer.waitFor({ state: 'visible', timeout: 10_000 })
await composer.click()
@@ -149,29 +171,45 @@ test.describe('sidebar states — subagent and background dot coexist', () => {
{ timeout: 15_000 },
)
// The background process (sleep 5) should show a "Background task
// running" dot while the subagent is also running.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should appear while subagent runs' },
)
.toBeGreaterThan(0)
// Evidence: the background dot is visible while the subagent runs.
await page.screenshot({ path: 'test-results/bg-dot-while-subagent-runs.png' })
// Now wait for the final answer to appear.
// While the turn is busy the dot-state priority paints the session as
// "working" ('Session running') — that claim OUTRANKS 'background', so
// polling for the bg dot mid-turn races the turn length against the poll
// budget. Wait for the turn to END (final text + running dot cleared),
// then assert the background dot as a stable, sentinel-held state.
await page.waitForFunction(
(text) => (document.body.textContent ?? '').includes(text),
SIDEBAR_CROSS_TEXTS.finalText,
{ timeout: 90_000 },
)
await expect
.poll(
() => page.locator(`[aria-label="${SESSION_RUNNING_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'session running dot should disappear after turn completes' },
)
.toBe(0)
// After the turn + auto-dismiss, the background dot should be gone.
await page.waitForTimeout(8000)
const bgCount = await page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count()
expect(bgCount, 'background dot should be gone after process exits').toBe(0)
// The background process is held open by the sentinel, so the bg dot is
// a stable state — poll only to absorb the event-driven flip landing a
// tick after the running dot clears.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should be visible after turn completes' },
)
.toBeGreaterThan(0)
// Evidence: the background dot is visible while the process runs.
await page.screenshot({ path: 'test-results/bg-dot-while-subagent-runs.png' })
// Release the process; the dot should clear on the completion event —
// event-driven, not a fixed sleep.
bgRelease.release()
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should be gone after process exits' },
)
.toBe(0)
})
})
@@ -190,6 +228,7 @@ test.describe('sidebar states — cross-session dot transition', () => {
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend({
extraConfig: DISABLE_AUTO_TITLE,
mockServer: { backgroundReleasePath: bgRelease.path },
})
await waitForAppReady(fixture, 120_000)
@@ -213,14 +252,13 @@ test.describe('sidebar states — cross-session dot transition', () => {
await composer.type('E2E_SIDEBAR_CROSS', { delay: 20 })
await page.keyboard.press('Enter')
// Wait for the background dot to appear.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should appear' },
)
.toBeGreaterThan(0)
// While the turn is busy the dot-state priority paints the session as
// "working" ('Session running') — that claim OUTRANKS 'background', so
// polling for the bg dot mid-turn races the turn length (two model trips
// + a real subagent spawn) against the poll budget: the CI flake this
// spec had. Wait for the turn to END first, then assert the bg dot as a
// stable, sentinel-held state.
//
// The final answer text streams before message.complete, so text visibility
// alone is not a completion barrier. Wait for the foreground-running state
// to clear before asserting the background-process state.
@@ -236,11 +274,16 @@ test.describe('sidebar states — cross-session dot transition', () => {
)
.toBe(0)
// The background dot must still be visible: the turn is done but the
// The background dot must be visible now: the turn is done but the
// process is held open by the sentinel, so this is a stable state rather
// than a window we have to catch in time.
const bgDuringTurn = await page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count()
expect(bgDuringTurn, 'background dot should still be visible after turn completes').toBeGreaterThan(0)
// than a window we have to catch in time. Poll to absorb the event-driven
// flip landing a tick after the running dot clears.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should be visible after turn completes' },
)
.toBeGreaterThan(0)
// Evidence: bg dot visible on session A while its turn is done but the
// background process hasn't exited yet.
+38 -14
View File
@@ -36,6 +36,18 @@ const BG_DOT_LABEL = 'Background task running'
/** Foreground turn-running dot aria-label. */
const SESSION_RUNNING_DOT_LABEL = 'Session running'
/**
* The auto-title auxiliary call hits the SAME mock provider as the chat turn,
* and its request carries the user's message — trigger keyword included. The
* mock's trigger matching is text-based, so the title call consumes a script
* index: the real chat turn then gets turn 2 (final answer, NO tool calls),
* the background process is never spawned, and the bg dot never appears.
* Whether that happens depends on which request lands first — the CI flake
* this spec had. Disable auto-title so script indices line up with real chat
* turns (same fix as interim-messages.spec.ts).
*/
const DISABLE_AUTO_TITLE = 'auxiliary:\n title_generation:\n enabled: false'
/** Locate a session's sidebar row by its preview text. */
function sessionRow(page: import('@playwright/test').Page, text: string) {
return page.locator('[data-slot="sidebar"] button').filter({ hasText: text }).first()
@@ -59,13 +71,13 @@ async function startTurnAndSwitchAway(page: import('@playwright/test').Page) {
{ timeout: 15_000 },
)
// Wait for the background dot — confirms the turn is running.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should appear' },
)
.toBeGreaterThan(0)
// NOTE: while the turn is busy the dot-state priority paints the session as
// "working" ('Session running'), which OUTRANKS the background claim — the
// 'Background task running' dot only appears once the turn completes while
// the (sentinel-held) process is still alive. Polling for the bg dot mid-turn
// races the turn length (two model trips + a real subagent spawn) against
// the poll budget, which is exactly the flake this spec had on CI. So: wait
// for the turn to END first, then assert the bg dot as a stable state.
// The final answer text streams before message.complete, so text visibility
// alone is not a completion barrier. Wait for the foreground-running state
@@ -82,11 +94,17 @@ async function startTurnAndSwitchAway(page: import('@playwright/test').Page) {
)
.toBe(0)
// The background dot must still be visible: the turn is done but the
// process is held open by the sentinel, so this is a stable state rather
// than a window we have to catch in time.
const bgDuringTurn = await page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count()
expect(bgDuringTurn, 'background dot should still be visible after turn completes').toBeGreaterThan(0)
// The background dot must be visible now: the turn is done but the process
// is held open by the sentinel, so this is a stable state rather than a
// window we have to catch in time. Poll rather than sampling once — the
// dot flip is event-driven off the busy=false publish and can land a tick
// after the running dot clears.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should be visible after turn completes' },
)
.toBeGreaterThan(0)
// Switch to a new session — session A is no longer $selectedStoredSessionId.
// This is required: openSessionTile bails if the session is already selected.
@@ -121,6 +139,7 @@ test.describe('sidebar states — tab (hidden) unread is correct', () => {
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend({
extraConfig: DISABLE_AUTO_TITLE,
mockServer: { backgroundReleasePath: bgRelease.path },
})
await waitForAppReady(fixture, 120_000)
@@ -142,7 +161,10 @@ test.describe('sidebar states — tab (hidden) unread is correct', () => {
// ⌃-click opens the session as a TAB (center dock = stacked, not visible
// unless it's the active tab). The session is NOT on screen.
const row = sessionRow(page, SIDEBAR_CROSS_TEXTS.finalText)
//
// With auto-title disabled the sidebar row is titled by the user's
// message (the trigger keyword), not the assistant's final text.
const row = sessionRow(page, 'E2E_SIDEBAR_CROSS')
await row.click({ modifiers: ['Control'] })
await page.waitForTimeout(2000)
@@ -182,6 +204,7 @@ test.describe.skip('sidebar states — split (visible) unread bug (RED)', () =>
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend({
extraConfig: DISABLE_AUTO_TITLE,
mockServer: { backgroundReleasePath: bgRelease.path },
})
await waitForAppReady(fixture, 120_000)
@@ -204,7 +227,8 @@ test.describe.skip('sidebar states — split (visible) unread bug (RED)', () =>
// Drag the session row from the sidebar to the right edge of the workspace
// zone to create a SPLIT (side-by-side) tile. This triggers the real
// startSessionDrag → onCommit → openSessionTile(id, 'right', anchor) path.
const row = sessionRow(page, SIDEBAR_CROSS_TEXTS.finalText)
// With auto-title disabled the sidebar row is titled by the user's message.
const row = sessionRow(page, 'E2E_SIDEBAR_CROSS')
const rowBox = await row.boundingBox()
expect(rowBox, 'session row must be visible').not.toBeNull()
+21 -4
View File
@@ -10,7 +10,7 @@
* `syncSessionStateToView` to fire a second `setMessages` — a visual
* flicker as the transcript DOM was updated.
*
* This test pre-seeds a 32-message session into state.db, boots the app,
* This test pre-seeds a session into state.db, boots the app,
* clicks the session (cold resume — populates the warm cache), navigates
* away to a new chat, then clicks back (warm resume). Two detectors run:
*
@@ -50,8 +50,16 @@ const SESSION_TITLE = 'E2E Warm Resume Jitter Test'
// renderer's keep-alive visibility policy instead of relying on DOM order.
const SURFACE = '[data-composer-target]:not([data-pane-hidden] [data-composer-target])'
const ALL_SURFACES = '[data-composer-target]'
/** 32 messages (16 user/assistant pairs) — enough DOM churn for detection. */
const MESSAGE_COUNT = 32
/**
* 16 messages (8 user/assistant pairs) — enough DOM churn for detection while
* still fitting a hot-hidden pane's retention budget. A kept-alive pane keeps
* only its live tail (HIDDEN_TRANSCRIPT_RENDER_BUDGET = 40 weight units in
* thread/list.tsx); 16 short messages ≈ 32 units, so the whole transcript
* survives hiding. Above the budget, reveal legitimately backfills trimmed
* turns (additive DOM bursts) — that is paging, not the repaint bug this
* suite hunts, and it would drown the detectors.
*/
const MESSAGE_COUNT = 16
/** Seeded PRNG so the generated content is deterministic across runs. */
const RNG_SEED = 42
@@ -174,7 +182,16 @@ async function installRenderCounter(
: surfaces.at(-1)
const viewport = surface?.querySelector('[data-slot="aui_thread-viewport"]')
if (!viewport) {
throw new Error('Thread viewport not found before warm resume')
const diag = [...document.querySelectorAll(allSelector)].map(s => ({
hidden: Boolean(s.closest('[data-pane-hidden]')),
target: s.getAttribute('data-composer-target'),
hasViewport: Boolean(s.querySelector('[data-slot="aui_thread-viewport"]')),
textLen: (s.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').length,
head: (s.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').slice(0, 80),
tail: (s.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').slice(-80),
includesExpected: expected ? (s.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').includes(expected) : null,
}))
throw new Error('Thread viewport not found before warm resume DIAG=' + JSON.stringify(diag) + ' expected=' + expected)
}
const state = { bursts: 0, mutations: 0, timeline: [] as number[], stopped: false, reconciles: 0 }
+247 -14
View File
@@ -1,6 +1,11 @@
import { describe, expect, it } from 'vitest'
import { execFileSync } from 'node:child_process'
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { detectBundleSkew, isFallbackCommit, type RunGit } from './bundle-skew'
import { afterAll, describe, expect, it } from 'vitest'
import { detectBundleSkew, isFallbackCommit, type RunGit, RUNTIME_PATHS } from './bundle-skew'
const REPO = '/repo'
const STAMP = { commit: 'a'.repeat(40), source: 'ci' }
@@ -9,6 +14,36 @@ function gitReturning(stdout: string, code = 0): RunGit {
return async () => ({ code, stderr: '', stdout })
}
/**
* A git fake that answers per subcommand, so a test can say "ancestry fails,
* but the count would have claimed skew" — which is the shape of #92233.
*/
function gitAnswering(answers: Record<string, { code?: number; stderr?: string; stdout?: string }>): {
calls: string[][]
git: RunGit
} {
const calls: string[][] = []
const git: RunGit = async args => {
calls.push(args)
const answer = answers[args[0]] ?? {}
return {
code: answer.code ?? 0,
stderr: answer.stderr ?? '',
stdout: answer.stdout ?? ''
}
}
return { calls, git }
}
/** Every subcommand succeeds; rev-list reports `count`. */
function gitCounting(count: string): RunGit {
return gitAnswering({ 'merge-base': { code: 0 }, 'rev-list': { stdout: count } }).git
}
describe('isFallbackCommit', () => {
it('matches the all-zero placeholder at any stamp length', () => {
expect(isFallbackCommit('0'.repeat(40))).toBe(true)
@@ -19,27 +54,21 @@ describe('isFallbackCommit', () => {
describe('detectBundleSkew', () => {
it('reports stale when desktop commits landed after the stamp', async () => {
const result = await detectBundleSkew(STAMP, gitReturning('3\n'), REPO)
const result = await detectBundleSkew(STAMP, gitCounting('3\n'), REPO)
expect(result).toEqual({ desktopCommitsBehind: 3, outOfSync: true })
})
it('passes the stamp range scoped to apps/desktop', async () => {
let seen: string[] = []
const git: RunGit = async args => {
seen = args
return { code: 0, stderr: '', stdout: '0' }
}
it('counts only commits that touch runtime desktop paths', async () => {
const { calls, git } = gitAnswering({ 'merge-base': { code: 0 }, 'rev-list': { stdout: '0' } })
await detectBundleSkew(STAMP, git, REPO)
expect(seen).toEqual(['rev-list', '--count', `${STAMP.commit}..HEAD`, '--', 'apps/desktop'])
expect(calls[1]).toEqual(['rev-list', '--count', `${STAMP.commit}..HEAD`, '--', ...RUNTIME_PATHS])
})
it('is quiet when no desktop commits follow the stamp', async () => {
const result = await detectBundleSkew(STAMP, gitReturning('0\n'), REPO)
const result = await detectBundleSkew(STAMP, gitCounting('0\n'), REPO)
expect(result).toEqual({ desktopCommitsBehind: 0, outOfSync: false })
})
@@ -79,9 +108,213 @@ describe('detectBundleSkew', () => {
})
it('is quiet on unparsable rev-list output', async () => {
expect(await detectBundleSkew(STAMP, gitReturning('fatal: bad object'), REPO)).toEqual({
expect(await detectBundleSkew(STAMP, gitCounting('fatal: bad object'), REPO)).toEqual({
desktopCommitsBehind: null,
outOfSync: false
})
})
// #92233: a ZIP-fallback update rewrites the tree into a synthetic root, so
// the stamp commit still RESOLVES but is unreachable from HEAD. `A..HEAD`
// then counts HEAD's own history instead of measuring skew, and reports a
// permanent 1 even though apps/desktop is byte-identical. The user gets an
// "App build out of date" warning that cannot go off, so no remedy clears it.
it('is quiet when the stamp is not an ancestor of HEAD', async () => {
const { git } = gitAnswering({
'merge-base': { code: 1 },
'rev-list': { stdout: '1\n' }
})
expect(await detectBundleSkew(STAMP, git, REPO)).toEqual({
desktopCommitsBehind: null,
outOfSync: false
})
})
it('does not consult the commit count once ancestry is refused', async () => {
const { calls, git } = gitAnswering({
'merge-base': { code: 1 },
'rev-list': { stdout: '9999\n' }
})
await detectBundleSkew(STAMP, git, REPO)
expect(calls.map(args => args[0])).toEqual(['merge-base'])
})
it('asks about ancestry before counting, against the same stamp', async () => {
const { calls, git } = gitAnswering({
'merge-base': { code: 0 },
'rev-list': { stdout: '2\n' }
})
const result = await detectBundleSkew(STAMP, git, REPO)
expect(calls[0]).toEqual(['merge-base', '--is-ancestor', STAMP.commit, 'HEAD'])
expect(calls[1]?.[0]).toBe('rev-list')
expect(result).toEqual({ desktopCommitsBehind: 2, outOfSync: true })
})
it('is quiet when git cannot answer the ancestry question at all', async () => {
const { git } = gitAnswering({
'merge-base': { code: 128 },
'rev-list': { stdout: '4\n' }
})
expect(await detectBundleSkew(STAMP, git, REPO)).toEqual({
desktopCommitsBehind: null,
outOfSync: false
})
})
// Shallow clones, measured against git 2.55 rather than assumed. A stamp
// commit from BEFORE the graft boundary is not an object the clone has, so
// `--is-ancestor` exits 128 with "Not a valid object name" — the same
// unknowable bucket as any other missing commit, not a shallow-specific
// failure. A stamp INSIDE the shallow graph is answered normally, so
// `--fetch-depth`-limited CI checkouts do not lose skew detection wholesale;
// only builds stamped deeper than the checkout goes do.
it('is quiet on a shallow clone whose stamp predates the graft boundary', async () => {
const { calls, git } = gitAnswering({
'merge-base': {
code: 128,
stderr: `fatal: Not a valid object name ${STAMP.commit}`
},
'rev-list': { stdout: '7\n' }
})
expect(await detectBundleSkew(STAMP, git, REPO)).toEqual({
desktopCommitsBehind: null,
outOfSync: false
})
expect(calls).toHaveLength(1)
})
it('still detects skew on a shallow clone when the stamp is in the graph', async () => {
const { git } = gitAnswering({
'merge-base': { code: 0 },
'rev-list': { stdout: '2\n' }
})
expect(await detectBundleSkew(STAMP, git, REPO)).toEqual({
desktopCommitsBehind: 2,
outOfSync: true
})
})
})
// Real-git integration: proves the pathspec discriminates docs/e2e-only
// commits from runtime commits, and that a disconnected stamp goes quiet, in
// an actual repository rather than against a hand-written fake.
const scratchRepos: string[] = []
afterAll(() => {
for (const dir of scratchRepos) {
rmSync(dir, { force: true, recursive: true })
}
})
function scratchGit(repoRoot: string) {
return (...args: string[]) =>
execFileSync('git', ['-c', 'user.email=skew@test', '-c', 'user.name=skew', ...args], {
cwd: repoRoot,
stdio: ['ignore', 'pipe', 'pipe']
})
.toString()
.trim()
}
function makeScratchRepo(): { base: string; repoRoot: string } {
const repoRoot = mkdtempSync(join(tmpdir(), 'bundle-skew-'))
scratchRepos.push(repoRoot)
const git = scratchGit(repoRoot)
git('init', '-q', '-b', 'main')
git('commit', '-q', '--allow-empty', '-m', 'base')
return { base: git('rev-parse', 'HEAD'), repoRoot }
}
function writeFiles(repoRoot: string, files: string[]) {
for (const file of files) {
const target = join(repoRoot, file)
mkdirSync(dirname(target), { recursive: true })
writeFileSync(target, '')
}
}
function realGitRun(root: string): RunGit {
return async (args, options) => {
try {
const stdout = execFileSync('git', args, {
cwd: options.cwd || root,
stdio: ['ignore', 'pipe', 'pipe']
}).toString()
return { code: 0, stderr: '', stdout }
} catch (error) {
const e = error as { status?: number; stderr?: Buffer; stdout?: Buffer }
return {
code: e.status ?? 1,
stderr: e.stderr?.toString() ?? '',
stdout: e.stdout?.toString() ?? ''
}
}
}
}
describe('detectBundleSkew against a real git repo', () => {
it('is quiet when only docs and e2e specs changed under apps/desktop', async () => {
const { base, repoRoot } = makeScratchRepo()
const git = scratchGit(repoRoot)
writeFiles(repoRoot, ['apps/desktop/AGENTS.md', 'apps/desktop/e2e/boot.spec.ts'])
git('add', '.')
git('commit', '-q', '-m', 'docs and e2e only')
const result = await detectBundleSkew({ commit: base, source: 'local' }, realGitRun(repoRoot), repoRoot)
expect(result).toEqual({ desktopCommitsBehind: 0, outOfSync: false })
})
it('warns when a renderer file changed under apps/desktop', async () => {
const { base, repoRoot } = makeScratchRepo()
const git = scratchGit(repoRoot)
writeFiles(repoRoot, ['apps/desktop/src/app/new-feature.tsx', 'apps/desktop/README.md'])
git('add', '.')
git('commit', '-q', '-m', 'renderer change')
const result = await detectBundleSkew({ commit: base, source: 'local' }, realGitRun(repoRoot), repoRoot)
expect(result).toEqual({ desktopCommitsBehind: 1, outOfSync: true })
})
// The #92233 install, reproduced: the update rewrote the tree onto a fresh
// orphan root, so the stamp resolves but is unreachable. Real git answers
// `rev-list` with a positive count here — ancestry is the only thing that
// keeps the banner off.
it('is quiet when the stamp sits on a disconnected root', async () => {
const { base, repoRoot } = makeScratchRepo()
const git = scratchGit(repoRoot)
git('checkout', '-q', '--orphan', 'rewritten')
writeFiles(repoRoot, ['apps/desktop/src/app/shell.tsx'])
git('add', '.')
git('commit', '-q', '-m', 'synthetic root after a ZIP-fallback update')
const runGit = realGitRun(repoRoot)
// Precondition: the raw count this function used to trust is nonzero.
const raw = await runGit(['rev-list', '--count', `${base}..HEAD`, '--', ...RUNTIME_PATHS], { cwd: repoRoot })
expect(Number.parseInt(raw.stdout.trim(), 10)).toBeGreaterThan(0)
const result = await detectBundleSkew({ commit: base, source: 'local' }, runGit, repoRoot)
expect(result).toEqual({ desktopCommitsBehind: null, outOfSync: false })
})
})
+60 -11
View File
@@ -10,20 +10,33 @@
* Bot Mode update" reports).
*
* Detection: the packaged build carries install-stamp.json with the commit
* it was built from. If commits touching `apps/desktop/` exist in the source
* tree AFTER that stamp commit, the running renderer is provably missing
* desktop changes the installed runtime has:
* it was built from. If commits touching the RUNTIME paths of apps/desktop
* exist in the source tree AFTER that stamp commit, the running renderer is
* provably missing desktop changes the installed runtime has:
*
* git rev-list --count <stampCommit>..HEAD -- apps/desktop
* git merge-base --is-ancestor <stampCommit> HEAD
* git rev-list --count <stampCommit>..HEAD -- <RUNTIME_PATHS>
*
* Scoping to `apps/desktop/` keeps this quiet for the common case where the
* repo advances with agent-only changes — a shell built before those is not
* stale in any way the user can see.
* Ancestry has to come first, because `A..HEAD` only means "how far HEAD is
* ahead of A" when A is an ancestor of HEAD. When it is not, the range
* degenerates to HEAD's own history and the count stops describing skew at
* all: an update that rewrote the tree into a synthetic root leaves a stamp
* commit that still resolves but sits on a disconnected graph, so the count
* is a permanent >= 1 even when apps/desktop is byte-identical (#92233).
* Resolving the stamp is not enough — an unknown commit already exits
* non-zero below, but a merely *unrelated* one exits 0 with a positive count.
*
* Scoping to runtime paths keeps this quiet for the common cases where the
* repo advances without user-visible desktop changes: agent-only commits
* elsewhere in the repo, and docs / e2e spec / dev-script churn under
* apps/desktop that never reaches the shipped renderer or main process
* (#99832).
*
* Fail-quiet by design: no stamp (dev runs), a fallback all-zero stamp
* (non-git build), an unknown commit (stamp predates a shallow clone's
* history), or any git failure all report "not stale". This warning must
* never false-positive — it tells users their install is torn.
* history), a stamp that is not an ancestor of HEAD, or any git failure all
* report "not stale". This warning must never false-positive — it tells
* users their install is torn.
*
* Pure + injectable so it is testable without booting Electron or git.
*/
@@ -35,7 +48,7 @@ export interface BundleSkewStamp {
}
export interface BundleSkewResult {
/** Commits under apps/desktop/ between the build stamp and HEAD (null = unknowable). */
/** Runtime-path commits between the build stamp and HEAD (null = unknowable). */
desktopCommitsBehind: null | number
/** True only on positive proof that the renderer predates desktop changes in the tree. */
outOfSync: boolean
@@ -46,6 +59,23 @@ export type RunGit = (
options: { cwd: string }
) => Promise<{ code: number; stderr: string; stdout: string }>
/**
* The apps/desktop paths that actually reach the user: renderer sources,
* main-process sources, the HTML entry, the public/ assets Vite copies into
* the bundle, app icons, and the packaging config. Docs, e2e specs, scratch
* scripts, and dev tooling never reach the shipped app, so a delta confined
* to them is not a torn install in any way the user can see.
*/
export const RUNTIME_PATHS = [
'apps/desktop/src',
'apps/desktop/electron',
'apps/desktop/index.html',
'apps/desktop/public',
'apps/desktop/assets',
'apps/desktop/package.json',
'apps/desktop/vite.config.ts'
] as const
const NOT_STALE: BundleSkewResult = { desktopCommitsBehind: null, outOfSync: false }
/** Matches write-build-stamp.mjs's all-zero placeholder for non-git builds. */
@@ -63,7 +93,26 @@ export async function detectBundleSkew(
}
try {
const result = await runGit(['rev-list', '--count', `${stamp.commit}..HEAD`, '--', 'apps/desktop'], {
// Exit 0 = ancestor, 1 = unrelated or diverged, anything else = git could
// not answer (unknown object, shallow clone, not a repo). Only the first
// makes the commit count below a statement about skew, and the other two
// are the same "unknowable" the branches above already answer quietly.
//
// Deliberately not falling back to comparing apps/desktop CONTENT here.
// Differing content would prove the build and the tree disagree, but not
// which way round: a user sitting on an older checkout than their build
// would be told "app build out of date" backwards. Ancestry is what makes
// this a proof that the renderer PREDATES the tree, which is the claim the
// warning actually makes.
const ancestry = await runGit(['merge-base', '--is-ancestor', stamp.commit, 'HEAD'], {
cwd: repoRoot
})
if (ancestry.code !== 0) {
return NOT_STALE
}
const result = await runGit(['rev-list', '--count', `${stamp.commit}..HEAD`, '--', ...RUNTIME_PATHS], {
cwd: repoRoot
})
+49
View File
@@ -0,0 +1,49 @@
import { describe, expect, it } from 'vitest'
import { detectBundleSwap } from './bundle-swap'
const RUNNING = { builtAt: '2026-08-29T04:00:00.000Z', commit: 'a'.repeat(40), source: 'local' }
describe('detectBundleSwap', () => {
it('reports a swap when the on-disk stamp carries a different commit', () => {
const onDisk = { ...RUNNING, commit: 'b'.repeat(40) }
expect(detectBundleSwap(RUNNING, onDisk)).toBe(true)
})
it('reports a swap when the same commit was rebuilt (builtAt moved)', () => {
const onDisk = { ...RUNNING, builtAt: '2026-08-31T23:55:41.149Z' }
expect(detectBundleSwap(RUNNING, onDisk)).toBe(true)
})
// The Windows locked-binary case (#92233): the swap leg failed, so the
// bundle on disk is still the one we are running. A relaunch would repair
// nothing and cost the user their window.
it('is quiet when the on-disk stamp matches the running one', () => {
expect(detectBundleSwap(RUNNING, { ...RUNNING })).toBe(false)
})
it('is quiet without a running stamp (dev runs)', () => {
expect(detectBundleSwap(null, { ...RUNNING })).toBe(false)
})
it('is quiet without an on-disk stamp (unreadable resources)', () => {
expect(detectBundleSwap(RUNNING, null)).toBe(false)
})
it('is quiet on a fallback stamp on either side (non-git build)', () => {
const fallbackTagged = { ...RUNNING, source: 'fallback' }
const fallbackCommit = { ...RUNNING, commit: '0'.repeat(40) }
expect(detectBundleSwap(fallbackTagged, { ...RUNNING, commit: 'b'.repeat(40) })).toBe(false)
expect(detectBundleSwap(RUNNING, fallbackCommit)).toBe(false)
})
it('treats a missing builtAt on either side as unprovable at the same commit', () => {
const noBuiltAt = { commit: RUNNING.commit, source: 'local' }
expect(detectBundleSwap(noBuiltAt, { ...RUNNING })).toBe(false)
expect(detectBundleSwap(RUNNING, noBuiltAt)).toBe(false)
})
})
+61
View File
@@ -0,0 +1,61 @@
/**
* Swapped-bundle detection.
*
* The detached updater (scripts/desktop-update/posix.sh mac_swap /
* windows.ps1) rebuilds and swaps the packaged app on disk AFTER
* `hermes update` exits. An instance that was launched from the PRE-swap
* bundle — the user reopened Hermes mid-update, the #50238 gesture the boot
* gate exists for — would otherwise proceed to run the NEW runtime under the
* OLD renderer. The updater's own `open`/relaunch leg cannot rescue it: the
* single-instance lock turns that into a focus of the parked process, so no
* process ever loads the new build.
*
* That is the stale-renderer tail of a FULLY SUCCESSFUL update: the "App
* build out of date" banner appears right after the update, while the Updates
* card says "You're on the latest version" and so offers nothing that would
* clear it.
*
* Detection: compare the install stamp this process loaded at boot with the
* one on disk now. A different commit — or a different builtAt at the same
* commit (a dirty-tree or content-hash rebuild) — means the bundle under our
* feet is not the one we are running, and a plain relaunch loads it.
*
* Fail-quiet like bundle-skew: a missing stamp on either side (dev runs,
* unreadable resources) or a fallback all-zero commit reports "not swapped".
* This must never false-positive — a positive triggers an automatic relaunch.
*
* Pure so it is testable without booting Electron.
*/
import { isFallbackCommit } from './bundle-skew'
export interface BundleSwapStamp {
/** write-build-stamp.mjs build timestamp — differs on every rebuild. */
builtAt?: null | string
commit: string
/** write-build-stamp.mjs source tag — 'fallback' means the commit is fake. */
source?: null | string
}
/** True only on positive proof that the bundle on disk is not the running one. */
export function detectBundleSwap(running: BundleSwapStamp | null, onDisk: BundleSwapStamp | null): boolean {
if (!running?.commit || !onDisk?.commit) {
return false
}
if (running.source === 'fallback' || isFallbackCommit(running.commit)) {
return false
}
if (onDisk.source === 'fallback' || isFallbackCommit(onDisk.commit)) {
return false
}
if (running.commit !== onDisk.commit) {
return true
}
// Same commit: only a builtAt PRESENT ON BOTH sides can prove a rebuild —
// a missing timestamp (older stamp schema) proves nothing.
return Boolean(running.builtAt && onDisk.builtAt && running.builtAt !== onDisk.builtAt)
}
@@ -697,10 +697,23 @@ test('registry local route: v1 REMOTE global mode forces a genuinely-local backe
assert.notEqual(route.poolKey, backendScopeKey(LOCAL_CONNECTION_ID, 'default'))
})
test('registry local route: a per-profile remote override also forces local', () => {
test('registry local route: a per-profile remote override delegates to the override (#90477)', () => {
// The per-profile SSH/remote override is the authoritative route for that
// profile. Forcing local here made the roster list the profile via its
// override but open the thread in a local child — which fails when the
// profile exists only on the remote. The override must win.
const route = resolveRegistryLocalRoute('research', { profileRemoteOverride: true })
assert.deepEqual(route, { delegate: false, poolKey: 'conn:local::research' })
assert.deepEqual(route, { delegate: true, poolKey: 'research' })
})
test('registry local route: per-profile override wins when global remote is also active', () => {
const route = resolveRegistryLocalRoute('research', {
globalRemote: true,
profileRemoteOverride: true
})
assert.deepEqual(route, { delegate: true, poolKey: 'research' })
})
// --- shouldDeferLocalEnumeration (roster's connect-on-demand for 'local') ---
+11 -1
View File
@@ -493,7 +493,17 @@ export function resolveRegistryLocalRoute(
): RegistryLocalRoute {
const profileKey = String(profile ?? '').trim() || 'default'
if (opts.globalRemote || opts.profileRemoteOverride) {
// A per-profile SSH/remote override is an explicit per-profile routing
// decision: the override owns this profile's backend, so the 'local' entry
// must delegate to the legacy profile route (which resolves the override),
// not spawn a forced-local child. Forcing local here is the #90477 split:
// the roster lists the profile via its override, but opening the thread
// spawned a local backend that fails when the profile doesn't exist locally.
if (opts.profileRemoteOverride) {
return { delegate: true, poolKey: profileKey }
}
if (opts.globalRemote) {
return { delegate: false, poolKey: `${backendScopePrefix(LOCAL_CONNECTION_ID)}${profileKey}` }
}
+101 -5
View File
@@ -88,6 +88,7 @@ import {
buildBrowserWindowUrl
} from './browser-windows'
import { detectBundleSkew } from './bundle-skew'
import { detectBundleSwap } from './bundle-swap'
import { applyConnectionChange, sshQuitShouldBlock, teardownSshState } from './connection-apply'
import {
apiRequestRegistryConnectionId,
@@ -2156,6 +2157,59 @@ function updateGateDeps() {
}
}
// One-shot guard for the automatic bundle-swap relaunch below: the relaunched
// instance carries this flag so a stamp that still mismatches (unreadable
// resources, exotic packaging) can never produce a relaunch loop.
const BUNDLE_SWAP_RELAUNCH_FLAG = '--hermes-bundle-swap-relaunched'
// How long the parked instance waits for its own scheduled exit to land before
// giving up and booting the stale build anyway. Better a torn renderer with a
// banner than a window that never comes back.
const BUNDLE_SWAP_RELAUNCH_FAILSAFE_MS = 15_000
// The detached updater swaps the packaged bundle on disk AFTER `hermes update`
// exits (posix.sh mac_swap / windows.ps1). An instance reopened mid-update —
// the #50238 gesture the gate above exists for — was launched from the
// PRE-swap bundle, and the updater's `open` leg then merely focuses us (single
// instance), so no process ever loads the new build. Letting boot proceed here
// runs the new runtime under the old renderer: exactly the skew
// detectRendererSkew() warns about, except the Updates card already says
// "latest", so the warning's own remedy has nothing to run.
//
// This is the earliest point where the swap is PROVABLE — it happens while we
// are parked on the gate, so checking any sooner (at `ready`, before the gate)
// only ever compares a stamp with itself. Relaunching here also keeps the
// boot-progress window up for the whole wait instead of leaving the user with
// no window at all.
//
// Returns true when the relaunch was scheduled; the caller must park rather
// than continue booting, because the process exits underneath it.
function relaunchIntoSwappedBundle() {
if (!IS_PACKAGED || process.argv.includes(BUNDLE_SWAP_RELAUNCH_FLAG)) {
return false
}
if (!detectBundleSwap(INSTALL_STAMP, loadInstallStamp())) {
return false
}
rememberLog('[updates] app bundle was swapped during the update; relaunching into the new build')
try {
app.relaunch({
args: [...buildNoSandboxRelaunchArgs(process.argv.slice(1)), BUNDLE_SWAP_RELAUNCH_FLAG]
})
} catch (err) {
rememberLog(`[updates] bundle-swap relaunch failed: ${err?.message || err}; continuing with the current build`)
return false
}
void exitAfterBackendShutdown(0)
return true
}
// Block until no live update is in progress (or we hit the wait timeout).
// Emits a boot-progress phase so the renderer shows "Update in progress…"
// rather than a frozen splash. Returns true if it parked at all.
@@ -2218,6 +2272,14 @@ async function waitForUpdateToFinish() {
if (outcome === 'timeout') {
rememberLog('[updates] update still in progress after wait timeout; starting backend anyway')
} else if (relaunchIntoSwappedBundle()) {
await advanceBootProgress('backend.update-restart', 'Restarting Hermes to load the updated app…', 14)
// Park while the scheduled exit lands so this stale build never starts a
// backend; the failsafe below only runs if the exit somehow does not.
await new Promise(resolve => setTimeout(resolve, BUNDLE_SWAP_RELAUNCH_FAILSAFE_MS))
rememberLog(
`[updates] relaunch did not land within ${BUNDLE_SWAP_RELAUNCH_FAILSAFE_MS}ms; continuing with the current build`
)
} else {
rememberLog('[updates] update finished; proceeding with backend start')
}
@@ -9460,6 +9522,7 @@ function isHermesProcess(pid) {
function migrateActiveProfileIfMissing() {
migrateActiveProfileIfMissingPure(DESKTOP_PROFILE_CONFIG_PATH, {
legacyActivePath: path.join(HERMES_HOME, 'active_profile'),
hermesHome: HERMES_HOME,
profilesRoot: path.join(HERMES_HOME, 'profiles'),
existsSync: p => fs.existsSync(p),
readFileSync: (p, enc) => fs.readFileSync(p, enc),
@@ -12990,7 +13053,11 @@ function focusWindow(win) {
win.focus()
}
function spawnSecondaryWindow({ sessionId, watch }: { sessionId?: string; watch?: boolean } = {}) {
function spawnSecondaryWindow({
sessionId,
profile,
watch
}: { sessionId?: string; profile?: null | string; watch?: boolean } = {}) {
const icon = getAppIconPath()
const win = new BrowserWindow({
@@ -13052,6 +13119,7 @@ function spawnSecondaryWindow({ sessionId, watch }: { sessionId?: string; watch?
win,
buildSessionWindowUrl(sessionId, {
devServer: DEV_SERVER,
profile,
rendererIndexPath: DEV_SERVER ? undefined : resolveRendererIndex(),
watch
}),
@@ -13062,8 +13130,8 @@ function spawnSecondaryWindow({ sessionId, watch }: { sessionId?: string; watch?
}
// Open (or focus) a standalone window for a single chat session.
function createSessionWindow(sessionId, { watch = false } = {}) {
return sessionWindows.openOrFocus(sessionId, () => spawnSecondaryWindow({ sessionId, watch }))
function createSessionWindow(sessionId, { profile = null, watch = false } = {}) {
return sessionWindows.openOrFocus(sessionId, () => spawnSecondaryWindow({ sessionId, profile, watch }))
}
// Popped-out in-app Browser: same webview + address bar as a docked Browser
@@ -14550,7 +14618,10 @@ ipcMain.handle('hermes:window:openSession', async (_event, sessionId, opts) => {
return { ok: false, error: 'invalid-session-id' }
}
createSessionWindow(sessionId.trim(), { watch: opts?.watch === true })
createSessionWindow(sessionId.trim(), {
profile: typeof opts?.profile === 'string' ? opts.profile : null,
watch: opts?.watch === true
})
return { ok: true }
})
@@ -16740,6 +16811,15 @@ ipcMain.on('hermes:translucency:support', event => {
event.returnValue = { glass: GLASS_SUPPORTED, translucency: TRANSLUCENCY_SUPPORTED }
})
// Launch-flag facts the renderer needs before first paint (same sendSync
// pattern as translucency). `--local` gates every local-models GUI surface;
// it arrives from `hermes desktop --local` or directly on Hermes.exe (a
// shortcut edit), and survives self-relaunches because collectRelaunchArgs
// only strips internal flags.
ipcMain.on('hermes:launch-flags', event => {
event.returnValue = { localModels: process.argv.includes('--local') }
})
ipcMain.on('hermes:translucency', (_event, payload) => {
const next = normalizeTranslucency(payload, GLASS_SUPPORTED)
const previous = translucencyState
@@ -17169,10 +17249,26 @@ ipcMain.handle('hermes:version', async () => {
platform: process.platform,
hermesRoot: resolveUpdateRoot(),
bundleOutOfSync: skew.outOfSync,
bundleCommitsBehind: skew.desktopCommitsBehind
bundleCommitsBehind: skew.desktopCommitsBehind,
// True when the bundle on disk is not the one this process loaded — a
// plain app restart (no rebuild, no installer) clears the skew above.
// Packaged only: a dev `--build-only` rewrites build/install-stamp.json
// under a running `npm start`, which is a rebuild the developer asked for,
// not a torn install to offer a restart for.
bundleSwapPending: IS_PACKAGED && detectBundleSwap(INSTALL_STAMP, loadInstallStamp())
}
})
// The About page's "Restart Hermes" button (shown when bundleSwapPending):
// load the already-swapped bundle without asking the user to quit manually.
// app.relaunch() re-executes by path, so the fresh process picks up whatever
// bundle now lives there.
ipcMain.handle('hermes:app:relaunch', async () => {
rememberLog('[updates] renderer requested an app relaunch (swapped bundle pending)')
app.relaunch({ args: buildNoSandboxRelaunchArgs(process.argv.slice(1)) })
void exitAfterBackendShutdown(0)
})
// ===========================================================================
// Uninstall — remove the Chat GUI (and optionally the agent / user data).
// ===========================================================================
+5
View File
@@ -10,10 +10,14 @@ import { contextBridge, ipcRenderer, webFrame, webUtils } from 'electron'
const translucencySupport = ipcRenderer.sendSync('hermes:translucency:support')
const hudWindowing = ipcRenderer.sendSync('hermes:hud:windowing')
const hudNativeDrag = hudWindowing?.nativeDrag === true
const launchFlags = ipcRenderer.sendSync('hermes:launch-flags')
contextBridge.exposeInMainWorld('hermesDesktop', {
glassSupported: translucencySupport?.glass === true,
translucencySupported: translucencySupport?.translucency === true,
// Launch-flag fact: the app was started with --local, so the renderer may
// show the local-models surfaces. Static for the window's lifetime.
localModelsEnabled: launchFlags?.localModels === true,
getConnection: profile => ipcRenderer.invoke('hermes:connection', profile),
// Registry-scoped backend resolution: { connectionId, profile } → descriptor.
getConnectionFor: payload => ipcRenderer.invoke('hermes:connection:for', payload),
@@ -487,6 +491,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
return () => ipcRenderer.removeListener('hermes:bootstrap:event', listener)
},
getVersion: () => ipcRenderer.invoke('hermes:version'),
relaunchApp: () => ipcRenderer.invoke('hermes:app:relaunch'),
getRemoteDisplayReason: () => ipcRenderer.invoke('hermes:get-remote-display-reason'),
uninstall: {
summary: () => ipcRenderer.invoke('hermes:uninstall:summary'),
+185 -6
View File
@@ -25,8 +25,12 @@ import {
listProfileDirs,
migrateActiveProfileIfMissing,
PROFILE_SCORE_MIN_SIZE_BYTES,
profileGatewayPidPath,
profileStateDbPath,
readExistingPreference,
readLegacyActiveProfile,
scoreStateDb
scoreStateDb,
withDefaultCandidate
} from './profile-migration'
// ---------------------------------------------------------------------------
@@ -136,6 +140,7 @@ function baseDeps(overrides: Record<string, unknown> = {}) {
return {
legacyActivePath: '/home/u/.hermes/active_profile',
hermesHome: '/home/u/.hermes',
profilesRoot: '/home/u/.hermes/profiles',
existsSync: fs.existsSync,
readFileSync: fs.readFileSync,
@@ -377,10 +382,11 @@ test('decideMigration returns null when no candidate scores and legacy is invali
test('decideMigration suppresses write when best is default (single-profile fallback)', () => {
// The whole point of the migration is to migrate AWAY from default when a
// better candidate exists. If 'default' wins the score, the install is
// single-profile and we leave it alone.
// default-primary and we leave it alone. Default's DB is $HERMES_HOME/state.db,
// not profiles/default/state.db.
const deps = baseDeps()
const d = decideMigration(null, [], ['default', 'coder'], deps, p => (p.endsWith('/default/state.db') ? 99 : 50))
const d = decideMigration(null, [], ['default', 'coder'], deps, p => (p.endsWith('/.hermes/state.db') ? 99 : 50))
assert.equal(d, null)
})
@@ -397,11 +403,20 @@ test('decideMigration still flags _migrated when legacy is invalid (undefined) b
// migrateActiveProfileIfMissing (orchestrator)
// ---------------------------------------------------------------------------
test('migrateActiveProfileIfMissing is a no-op when the preference file exists', () => {
test('migrateActiveProfileIfMissing is a no-op when a user-selected preference file exists', () => {
// No `_migrated` flag = explicit user/CLI choice. Even a huge other profile
// must not steal the pin.
let written: unknown = null
const fs = makeFs({
'/cfg/active-profile.json': { content: '{"profile":"coder"}' },
'/home/u/.hermes/profiles/coder': { dir: true },
'/home/u/.hermes/profiles/writer': { dir: true },
'/home/u/.hermes/profiles/writer/state.db': { size: 400 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
existsSync: (p: string) => p === '/cfg/active-profile.json',
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
@@ -453,10 +468,11 @@ test('migrateActiveProfileIfMissing writes heuristic choice with _migrated=true'
test('migrateActiveProfileIfMissing is a no-op for single-profile (default-only) installs', () => {
// No heuristic candidate can beat 'default', so the orchestrator must NOT
// write a file — preserves legacy launch behavior for the 99% case.
// Production default DB is ~/.hermes/state.db, not profiles/default/state.db.
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/profiles/default/state.db': { size: 10 * 1024 * 1024, mtime: NOW - 86_400_000 }
'/home/u/.hermes/state.db': { size: 10 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
@@ -506,3 +522,166 @@ test('migrateActiveProfileIfMissing prefers a single running gateway over heuris
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), true)
assert.deepEqual(written, { profile: 'coder' })
})
// ---------------------------------------------------------------------------
// Production layout: default is ~/.hermes, not ~/.hermes/profiles/default
// ---------------------------------------------------------------------------
test('profileStateDbPath puts default at hermesHome, named under profilesRoot', () => {
assert.equal(profileStateDbPath('default', '/home/u/.hermes', '/home/u/.hermes/profiles'), '/home/u/.hermes/state.db')
assert.equal(
profileStateDbPath('conduit', '/home/u/.hermes', '/home/u/.hermes/profiles'),
'/home/u/.hermes/profiles/conduit/state.db'
)
})
test('profileGatewayPidPath puts default at hermesHome', () => {
assert.equal(
profileGatewayPidPath('default', '/home/u/.hermes', '/home/u/.hermes/profiles'),
'/home/u/.hermes/gateway.pid'
)
assert.equal(
profileGatewayPidPath('coder', '/home/u/.hermes', '/home/u/.hermes/profiles'),
'/home/u/.hermes/profiles/coder/gateway.pid'
)
})
test('withDefaultCandidate always leads with default and dedupes', () => {
assert.deepEqual(withDefaultCandidate([]), ['default'])
assert.deepEqual(withDefaultCandidate(['conduit']), ['default', 'conduit'])
assert.deepEqual(withDefaultCandidate(['default', 'conduit']), ['default', 'conduit'])
})
test('findRunningGatewayProfiles sees default gateway.pid at hermesHome', () => {
const fs = makeFs({
'/home/u/.hermes/gateway.pid': { content: '{"pid":99}' },
'/home/u/.hermes/profiles/coder/gateway.pid': { content: '{"pid":11}' }
})
assert.deepEqual(
findRunningGatewayProfiles('/home/u/.hermes/profiles', ['default', 'coder'], {
...fs,
hermesHome: '/home/u/.hermes',
isHermesProcess: pid => pid === 99
}),
['default']
)
})
test('migrateActiveProfileIfMissing does not pin a tiny named profile over a large default DB', () => {
// Regression for #100576: first-boot after update listed only
// ~/.hermes/profiles/<name>, never scored ~/.hermes/state.db, and wrote
// { profile: named, _migrated: true }.
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/profiles/conduit': { dir: true },
'/home/u/.hermes/state.db': { size: 409 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/conduit/state.db': { size: 2 * 1024 * 1024, mtime: NOW - 60_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
test('migrateActiveProfileIfMissing still pins a named profile that actually beats default', () => {
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/profiles/work': { dir: true },
'/home/u/.hermes/state.db': { size: 5 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/work/state.db': { size: 200 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), true)
assert.deepEqual(written, { profile: 'work', _migrated: true })
})
test('migrateActiveProfileIfMissing does not pin default when only default gateway is running', () => {
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/gateway.pid': { content: '{"pid":7}' },
'/home/u/.hermes/state.db': { size: 10 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
isHermesProcess: (pid: number) => pid === 7,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
test('readExistingPreference treats _migrated as heuristic-owned', () => {
const fs = makeFs({
'/cfg/active-profile.json': { content: '{"profile":"conduit","_migrated":true}' }
})
assert.deepEqual(readExistingPreference('/cfg/active-profile.json', fs.readFileSync), {
profile: 'conduit',
migrated: true
})
})
test('migrateActiveProfileIfMissing repairs a pre-existing heuristic pin when default now wins', () => {
// Sol P1 / #100576: file already exists with _migrated:true so first-boot
// skip left affected installs stuck. Re-score and clear.
let written: unknown = null
const fs = makeFs({
'/cfg/active-profile.json': { content: '{"profile":"conduit","_migrated":true}' },
'/home/u/.hermes/profiles/conduit': { dir: true },
'/home/u/.hermes/state.db': { size: 409 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/conduit/state.db': { size: 2 * 1024 * 1024, mtime: NOW - 60_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), true)
assert.deepEqual(written, { profile: null })
})
test('migrateActiveProfileIfMissing leaves a still-correct heuristic pin alone', () => {
let written: unknown = null
const fs = makeFs({
'/cfg/active-profile.json': { content: '{"profile":"work","_migrated":true}' },
'/home/u/.hermes/profiles/work': { dir: true },
'/home/u/.hermes/state.db': { size: 5 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/work/state.db': { size: 200 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
+100 -18
View File
@@ -15,6 +15,9 @@ export const PROFILE_SCORE_MIN_SIZE_BYTES = 1024
export interface MigrationDeps {
legacyActivePath: string
/** Default profile home (`~/.hermes`). Default's state.db and gateway.pid live here. */
hermesHome: string
/** Named-profile root (`~/.hermes/profiles`). Does not contain `default`. */
profilesRoot: string
existsSync: (path: string) => boolean
readFileSync: (path: string, encoding: 'utf8') => string
@@ -27,11 +30,38 @@ export interface MigrationDeps {
}
export interface MigrationDecision {
profile: string
profile: string | null
/** True when chosen from the state.db heuristic (auto-detected), undefined when explicit. */
_migrated?: boolean
}
/**
* Production layout: default IS `hermesHome`; named profiles are children of
* `profilesRoot`. There is no `profiles/default` directory on a normal install.
*/
export function profileStateDbPath(name: string, hermesHome: string, profilesRoot: string): string {
return name === 'default' ? `${hermesHome}/state.db` : `${profilesRoot}/${name}/state.db`
}
export function profileGatewayPidPath(name: string, hermesHome: string, profilesRoot: string): string {
return name === 'default' ? `${hermesHome}/gateway.pid` : `${profilesRoot}/${name}/gateway.pid`
}
function resolveHermesHome(profilesRoot: string, hermesHome?: string): string {
if (hermesHome) {
return hermesHome
}
// Tests that predate hermesHome pass only profilesRoot.
for (const suffix of ['/profiles', '\\profiles']) {
if (profilesRoot.endsWith(suffix)) {
return profilesRoot.slice(0, -suffix.length)
}
}
return profilesRoot
}
/**
* Parse the legacy CLI-sticky file. Returns the trimmed name on success, null when
* missing/unreadable/empty, undefined when present but invalid (so the caller can
@@ -70,16 +100,20 @@ export function readLegacyActiveProfile(
* Return the profile names whose gateway.pid file points to a live hermes process.
* Tolerates missing/malformed pid files and stale-but-recycled PIDs (the latter is
* the whole reason we check both liveness AND cmdline identity).
*
* `hermesHome` is optional so existing call sites that only pass `profilesRoot`
* still work: it is derived as the parent of `…/profiles`.
*/
export function findRunningGatewayProfiles(
profilesRoot: string,
allProfiles: string[],
deps: Pick<MigrationDeps, 'existsSync' | 'readFileSync' | 'isHermesProcess'>
deps: Pick<MigrationDeps, 'existsSync' | 'readFileSync' | 'isHermesProcess'> & { hermesHome?: string }
): string[] {
const hermesHome = resolveHermesHome(profilesRoot, deps.hermesHome)
const running: string[] = []
for (const name of allProfiles) {
const pidFile = `${profilesRoot}/${name}/gateway.pid`
const pidFile = profileGatewayPidPath(name, hermesHome, profilesRoot)
if (!deps.existsSync(pidFile)) {
continue
@@ -156,7 +190,7 @@ export function decideMigration(
let maxScore = -Infinity
for (const name of candidates) {
const s = score(`${deps.profilesRoot}/${name}/state.db`)
const s = score(profileStateDbPath(name, deps.hermesHome, deps.profilesRoot))
if (s == null) {
continue
@@ -176,8 +210,9 @@ export function decideMigration(
}
/**
* List known profile directory names under `profilesRoot`. Accepts `default` and
* any name passing the injected validator. Returns [] on missing dir or empty.
* List named profile directory names under `profilesRoot`. A directory named
* `default` is accepted if present (unusual) but production default is not a
* child of this folder — see `withDefaultCandidate`.
*/
export function listProfileDirs(deps: MigrationDeps): string[] {
let entries: Dirent[]
@@ -193,25 +228,59 @@ export function listProfileDirs(deps: MigrationDeps): string[] {
.map(e => e.name)
}
/** Default is always a candidate; it is `$HERMES_HOME`, not `$HERMES_HOME/profiles/default`. */
export function withDefaultCandidate(named: string[]): string[] {
return ['default', ...named.filter(name => name !== 'default')]
}
/**
* Orchestrator. Idempotent: writes at most once when the preference file is
* missing. Thin on top of the decision helpers above; the testable surface is
* `decideMigration` + the individual rung helpers, this function just glues them
* to the deps bag.
* Read an existing active-profile.json. Returns null when missing/malformed.
* `_migrated: true` means the first-boot heuristic wrote it (safe to re-score).
* Absence of that flag is a user/CLI choice and must not be overwritten.
*/
export function readExistingPreference(
desktopProfileConfigPath: string,
readFile: MigrationDeps['readFileSync']
): { profile: string | null; migrated: boolean } | null {
let parsed: unknown
try {
parsed = JSON.parse(readFile(desktopProfileConfigPath, 'utf8'))
} catch {
return null
}
if (!parsed || typeof parsed !== 'object') {
return null
}
const rec = parsed as { profile?: unknown; _migrated?: unknown }
const raw = typeof rec.profile === 'string' ? rec.profile.trim() : ''
return {
profile: raw || null,
migrated: rec._migrated === true
}
}
/**
* First-boot seed, plus repair of heuristic-owned files (`_migrated: true`).
* User-selected files (no `_migrated`) are never overwritten. When a repaired
* heuristic would now pick default, write `{ profile: null }` so Desktop drops
* `--profile` instead of pinning `default`.
*/
export function migrateActiveProfileIfMissing(desktopProfileConfigPath: string, deps: MigrationDeps): boolean {
if (deps.existsSync(desktopProfileConfigPath)) {
const existing = deps.existsSync(desktopProfileConfigPath)
? readExistingPreference(desktopProfileConfigPath, deps.readFileSync)
: null
if (existing && !existing.migrated) {
return false
}
const legacyActive = readLegacyActiveProfile(deps.legacyActivePath, deps.readFileSync, deps.isValidProfileName)
const allProfiles = listProfileDirs(deps)
if (allProfiles.length === 0) {
return false
}
const allProfiles = withDefaultCandidate(listProfileDirs(deps))
const running = findRunningGatewayProfiles(deps.profilesRoot, allProfiles, deps)
const candidates = running.length > 1 ? running : allProfiles
@@ -219,7 +288,20 @@ export function migrateActiveProfileIfMissing(desktopProfileConfigPath: string,
scoreStateDb(dbPath, deps.now(), deps.statSync)
)
if (!decision) {
// Same as the heuristic rung: pinning `default` into active-profile.json
// launches `hermes --profile default` and is worse than writing nothing
// (legacy sticky / implicit default). Covers a lone default gateway.pid.
if (!decision || decision.profile === 'default') {
if (existing?.migrated) {
deps.writeJson(desktopProfileConfigPath, { profile: null })
return true
}
return false
}
if (existing?.migrated && existing.profile === decision.profile) {
return false
}
@@ -68,6 +68,12 @@ test('buildSessionWindowUrl avoids a double slash when the dev server has a trai
assert.equal(url, 'http://localhost:5173/?win=secondary#/abc123')
})
test('buildSessionWindowUrl carries the owning profile in the query before the hash (#82768)', () => {
const url = buildSessionWindowUrl('abc123', { devServer: 'http://localhost:5173', profile: 'work', watch: true })
assert.equal(url, 'http://localhost:5173/?win=secondary&watch=1&profile=work#/abc123')
})
test('buildSessionWindowUrl encodes the session id in the hash route', () => {
const url = buildSessionWindowUrl('a b/c', { devServer: 'http://localhost:5173' })
+7 -2
View File
@@ -64,8 +64,13 @@ function chatWindowWebPreferences(preloadPath: string) {
// onboarding overlays and the global session sidebar. `watch=1` marks a
// spectator window (e.g. a running subagent's session): the renderer resumes it
// lazily so the gateway never builds an agent just to stream into it.
function buildSessionWindowUrl(sessionId: string, { devServer, rendererIndexPath, watch }: any = {}) {
const query = `?win=secondary${watch ? '&watch=1' : ''}`
// `profile` names the backend the window must boot against (same carry as the
// HUD's buildHudWindowUrl): without it a pop-out/watch window adopts the
// PRIMARY profile and resolves the session id against the wrong backend
// (#82768, #61286). Absent → unchanged primary adoption.
function buildSessionWindowUrl(sessionId: string, { devServer, profile, rendererIndexPath, watch }: any = {}) {
const profileKey = typeof profile === 'string' ? profile.trim() : ''
const query = `?win=secondary${watch ? '&watch=1' : ''}${profileKey ? `&profile=${encodeURIComponent(profileKey)}` : ''}`
const route = `#/${encodeURIComponent(sessionId)}`
if (devServer) {
+1 -1
View File
@@ -27,7 +27,7 @@
"profile:main": "tsc --build tsconfig.electron.json && wait-on http://127.0.0.1:5174 && node scripts/bundle-electron-main.mjs --dev && cross-env XCURSOR_SIZE=24 HERMES_DESKTOP_DEV_SERVER=http://127.0.0.1:5174 electron --inspect=9229 .",
"profile:main:cpu": "tsc --build tsconfig.electron.json && wait-on http://127.0.0.1:5174 && node scripts/bundle-electron-main.mjs --dev && cross-env XCURSOR_SIZE=24 NODE_OPTIONS=--cpu-prof HERMES_DESKTOP_DEV_SERVER=http://127.0.0.1:5174 electron .",
"start": "npm run build && electron .",
"prebuild": "npm run clean",
"prebuild": "node scripts/assert-root-install.mjs && npm run clean",
"build": "node scripts/assert-root-install.mjs && node scripts/write-build-stamp.mjs && vite build && node scripts/bundle-electron-main.mjs && node scripts/stage-native-deps.mjs",
"postbuild": "node scripts/assert-dist-built.mjs",
"prebuilder": "node scripts/patch-electron-builder-mac-binary.mjs",
+140 -29
View File
@@ -1,35 +1,146 @@
import { accessSync, readFileSync } from "fs"
// Build-time guard: refuse to start a build the installed tree cannot finish.
//
// The desktop workspace's dependencies are hoisted to the repo-root
// `node_modules`, so a root install that only covers *part* of the workspace
// graph leaves this app importable-looking but unbuildable. The guard exists to
// turn that into one actionable line ("run npm ci from the repo root") instead
// of a failure deep inside vite.
//
// It runs from `prebuild`, ahead of `npm run clean`, so a tree that cannot
// build is rejected before the build starts deleting its own outputs. `build`
// re-runs it for anyone invoking the build steps directly; the check is pure
// filesystem lookups, so paying for it twice costs nothing.
import { existsSync, readFileSync } from "fs"
import { createRequire } from "module"
import { resolve, join } from "path"
import { resolve, join, dirname } from "path"
import { isMain } from "./utils.mjs"
const app = resolve(import.meta.dirname, "..")
const root = resolve(app, "..", "..")
// Packages the build *consumes*, as opposed to merely declares. Each one is
// load-bearing for a distinct build step, and each one has been observed
// missing from a partial root install:
//
// vite — bundles the renderer (`vite build`).
// katex — `src/styles.css` imports `katex/dist/katex.min.css`, so
// the CSS transform fails before a single chunk is emitted.
// electron — the runtime electron-builder packages; without it `pack`
// cannot produce an unpacked app at all.
// electron-builder — the packager `npm run builder` shells out to.
//
// Checking only `vite` (the original guard) passes a tree missing any of the
// others, which is how an incomplete install reached `vite build` and died on
// an unresolved `katex/dist/katex.min.css` with no hint that the install — not
// the source — was at fault (#86443).
//
// These four are the documented floor — always checked, even when the app's
// package.json cannot be read. The full class is wider: EVERY non-optional
// package the workspace manifest declares is something the build may import
// (`vite.config.ts` pulls `@rolldown/plugin-babel`, `@vitejs/plugin-react`,
// `@tailwindcss/vite`; `bundle-electron-main.mjs` pulls `esbuild`; the renderer
// imports the rest). A hand-maintained list drifts the moment a new import
// lands, so `checkRootInstall` unions the floor with the manifest's declared
// `dependencies` + `devDependencies` — a partial install is refused whichever
// package it happened to drop. `optionalDependencies` are excluded by design:
// npm legitimately skips them (platform-gated natives like `get-windows`).
const BUILD_CRITICAL_PACKAGES = ["vite", "katex", "electron", "electron-builder"]
export { BUILD_CRITICAL_PACKAGES }
try {
accessSync(join(root, "node_modules", "vite", "package.json"))
} catch {
console.error(`Run from repo root: cd ${root} && npm ci`)
process.exit(1)
// Resolve the way Node's own lookup does — walk `node_modules` upward — rather
// than through `require.resolve`. A package whose `exports` map does not expose
// `./package.json` is not resolvable by path even when correctly installed, and
// that must not read as "missing". Scoped names (`@scope/name`) are a nested
// directory under `node_modules`, which `join` handles.
function packageIsInstalled(name, fromDir) {
let dir = fromDir
for (;;) {
if (existsSync(join(dir, "node_modules", name, "package.json"))) return true
const parent = dirname(dir)
if (parent === dir) return false
dir = parent
}
}
// `vite.config.ts` aliases react/react-dom to whatever this workspace resolves,
// and React refuses to run when the two come from different installed copies
// ("Minified React error #527" — it throws before the first paint, so the app
// window stays blank). npm stays silent about the split because the hoisted
// react still satisfies react-dom's caret peer range. Fail the build loudly
// instead of shipping a white screen.
const requireFromApp = createRequire(join(app, "package.json"))
const installedVersion = (pkg) =>
JSON.parse(readFileSync(requireFromApp.resolve(`${pkg}/package.json`), "utf8")).version
const react = installedVersion("react")
const reactDom = installedVersion("react-dom")
if (react !== reactDom) {
console.error(
`react@${react} / react-dom@${reactDom} version mismatch — React would fail ` +
`with error #527 and render a blank window. Pin both to the same version ` +
`in ${join(app, "package.json")}, then reinstall: cd ${root} && npm ci`
)
process.exit(1)
// Every package the workspace manifest at `appDir` declares as required
// (`dependencies` + `devDependencies`; never `optionalDependencies`). An
// unreadable or malformed manifest yields [] — the floor still applies, and
// the build's own manifest read fails loudly on its own.
export function requiredPackages(appDir) {
try {
const manifest = JSON.parse(readFileSync(join(appDir, "package.json"), "utf8"))
return [
...Object.keys(manifest.dependencies ?? {}),
...Object.keys(manifest.devDependencies ?? {}),
]
} catch {
return []
}
}
// Pure check — returns { ok: true } or { ok: false, error: "..." }.
// Kept side-effect-free so it can be unit tested without spawning a process.
export function checkRootInstall(appDir, rootDir) {
const wanted = [...new Set([...BUILD_CRITICAL_PACKAGES, ...requiredPackages(appDir)])]
const missing = wanted.filter(pkg => !packageIsInstalled(pkg, appDir))
if (missing.length > 0) {
return {
ok: false,
error:
`the desktop build needs ${missing.join(", ")}, which the current install ` +
`does not provide. A partial root install leaves the workspace looking ` +
`present while the build cannot complete. Reinstall from the repo root: ` +
`cd ${rootDir} && npm ci`
}
}
// `vite.config.ts` aliases react/react-dom to whatever this workspace resolves,
// and React refuses to run when the two come from different installed copies
// ("Minified React error #527" — it throws before the first paint, so the app
// window stays blank). npm stays silent about the split because the hoisted
// react still satisfies react-dom's caret peer range. Fail the build loudly
// instead of shipping a white screen.
const requireFromApp = createRequire(join(appDir, "package.json"))
const installedVersion = pkg =>
JSON.parse(readFileSync(requireFromApp.resolve(`${pkg}/package.json`), "utf8")).version
let react
let reactDom
try {
react = installedVersion("react")
reactDom = installedVersion("react-dom")
} catch (err) {
// Both are in BUILD_CRITICAL_PACKAGES' spirit but not its list: they are
// checked by version, and an unreadable package.json is a broken install
// rather than an absent one. Report it as such instead of throwing.
return {
ok: false,
error: `could not read the installed react/react-dom versions (${err.message}). Reinstall from the repo root: cd ${rootDir} && npm ci`
}
}
if (react !== reactDom) {
return {
ok: false,
error:
`react@${react} / react-dom@${reactDom} version mismatch — React would fail ` +
`with error #527 and render a blank window. Pin both to the same version ` +
`in ${join(appDir, "package.json")}, then reinstall: cd ${rootDir} && npm ci`
}
}
return { ok: true }
}
function main() {
const app = resolve(import.meta.dirname, "..")
const root = resolve(app, "..", "..")
const result = checkRootInstall(app, root)
if (!result.ok) {
console.error(`✗ assert-root-install: ${result.error}`)
process.exit(1)
}
}
if (isMain(import.meta.url)) {
main()
}
@@ -0,0 +1,197 @@
import assert from 'node:assert/strict'
import fs from 'node:fs'
import os from 'node:os'
import path from 'node:path'
import { test } from 'vitest'
import { BUILD_CRITICAL_PACKAGES as BUILD_CRITICAL, checkRootInstall, requiredPackages } from '../scripts/assert-root-install.mjs'
// Build a throwaway repo shaped like this one: an app workspace whose
// dependencies are hoisted to the repo root, which is what the guard walks.
// `manifest` is merged into the app's package.json so tests can declare
// dependencies the guard is expected to read.
function makeTree({ rootPackages = BUILD_CRITICAL, react = '19.2.7', reactDom = '19.2.7', manifest = {} } = {}) {
const tempRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-assert-root-'))
const appDir = path.join(tempRoot, 'apps', 'desktop')
fs.mkdirSync(appDir, { recursive: true })
fs.writeFileSync(path.join(appDir, 'package.json'), JSON.stringify({ name: 'desktop', ...manifest }), 'utf8')
const writePackage = (name, version) => {
const dir = path.join(tempRoot, 'node_modules', name)
fs.mkdirSync(dir, { recursive: true })
fs.writeFileSync(path.join(dir, 'package.json'), JSON.stringify({ name, version }), 'utf8')
}
for (const name of rootPackages) writePackage(name, '1.0.0')
if (react !== null) writePackage('react', react)
if (reactDom !== null) writePackage('react-dom', reactDom)
return { tempRoot, appDir }
}
test('checkRootInstall passes on a complete root install', () => {
const { tempRoot, appDir } = makeTree()
try {
assert.deepEqual(checkRootInstall(appDir, tempRoot), { ok: true })
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
// The regression this guard was widened for: the updater's partial `npm install`
// left katex out while vite was present, so the old vite-only check passed and
// the build died on an unresolved `katex/dist/katex.min.css` (#86443).
test('checkRootInstall fails when katex is missing but vite is present', () => {
const { tempRoot, appDir } = makeTree({
rootPackages: BUILD_CRITICAL.filter(name => name !== 'katex')
})
try {
const result = checkRootInstall(appDir, tempRoot)
assert.equal(result.ok, false)
assert.match(result.error, /katex/)
assert.match(result.error, /npm ci/)
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
test('checkRootInstall fails when electron is missing', () => {
const { tempRoot, appDir } = makeTree({
rootPackages: BUILD_CRITICAL.filter(name => name !== 'electron')
})
try {
const result = checkRootInstall(appDir, tempRoot)
assert.equal(result.ok, false)
assert.match(result.error, /electron/)
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
test('checkRootInstall reports every missing package at once', () => {
const { tempRoot, appDir } = makeTree({ rootPackages: ['vite'] })
try {
const result = checkRootInstall(appDir, tempRoot)
assert.equal(result.ok, false)
for (const name of ['katex', 'electron', 'electron-builder']) {
assert.match(result.error, new RegExp(name))
}
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
// The original guard's only check — kept, so widening coverage cannot silently
// drop the case it already handled.
test('checkRootInstall still fails when vite is missing', () => {
const { tempRoot, appDir } = makeTree({
rootPackages: BUILD_CRITICAL.filter(name => name !== 'vite')
})
try {
const result = checkRootInstall(appDir, tempRoot)
assert.equal(result.ok, false)
assert.match(result.error, /vite/)
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
test('checkRootInstall fails on a react/react-dom version split', () => {
const { tempRoot, appDir } = makeTree({ react: '19.2.7', reactDom: '19.1.0' })
try {
const result = checkRootInstall(appDir, tempRoot)
assert.equal(result.ok, false)
assert.match(result.error, /#527/)
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
// A package installed into the app's own node_modules rather than hoisted to the
// root is still installed. The guard walks upward like Node does, so it must not
// insist on the hoisted location.
test('checkRootInstall accepts a package nested in the app workspace', () => {
const { tempRoot, appDir } = makeTree({
rootPackages: BUILD_CRITICAL.filter(name => name !== 'katex')
})
const nested = path.join(appDir, 'node_modules', 'katex')
fs.mkdirSync(nested, { recursive: true })
fs.writeFileSync(path.join(nested, 'package.json'), JSON.stringify({ name: 'katex' }), 'utf8')
try {
assert.deepEqual(checkRootInstall(appDir, tempRoot), { ok: true })
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
// The class, not the four instances: the floor list is what a partial install
// has been *seen* to drop, but any declared non-optional package can be the one
// missing next (`vite.config.ts` imports `@rolldown/plugin-babel`, which the
// floor never named). The guard must read the manifest so the list cannot drift
// behind a new import.
test('checkRootInstall fails when a declared devDependency outside the floor is missing', () => {
const { tempRoot, appDir } = makeTree({
manifest: { devDependencies: { '@rolldown/plugin-babel': '1.0.0', esbuild: '1.0.0' } },
rootPackages: [...BUILD_CRITICAL, 'esbuild']
})
try {
const result = checkRootInstall(appDir, tempRoot)
assert.equal(result.ok, false)
assert.match(result.error, /@rolldown\/plugin-babel/)
assert.doesNotMatch(result.error, /esbuild/)
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
test('checkRootInstall fails when a declared runtime dependency is missing', () => {
const { tempRoot, appDir } = makeTree({
manifest: { dependencies: { '@vscode/codicons': '1.0.0' } }
})
try {
const result = checkRootInstall(appDir, tempRoot)
assert.equal(result.ok, false)
assert.match(result.error, /@vscode\/codicons/)
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
// npm skips optionalDependencies legitimately (platform-gated natives), so an
// absent optional package is not a partial install.
test('checkRootInstall ignores missing optionalDependencies', () => {
const { tempRoot, appDir } = makeTree({
manifest: { optionalDependencies: { 'get-windows': '9.3.0' } }
})
try {
assert.deepEqual(checkRootInstall(appDir, tempRoot), { ok: true })
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
test('checkRootInstall passes when every declared package is installed', () => {
const { tempRoot, appDir } = makeTree({
manifest: { dependencies: { '@scope/pkg': '1.0.0' }, devDependencies: { esbuild: '1.0.0' } },
rootPackages: [...BUILD_CRITICAL, '@scope/pkg', 'esbuild']
})
try {
assert.deepEqual(checkRootInstall(appDir, tempRoot), { ok: true })
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
// The floor is unconditional: a manifest the guard cannot parse must not turn
// the check off.
test('checkRootInstall keeps the floor when the manifest is unreadable', () => {
const { tempRoot, appDir } = makeTree({ rootPackages: ['vite'] })
fs.writeFileSync(path.join(appDir, 'package.json'), '{not json', 'utf8')
try {
assert.deepEqual(requiredPackages(appDir), [])
const result = checkRootInstall(appDir, tempRoot)
assert.equal(result.ok, false)
assert.match(result.error, /katex/)
} finally {
fs.rmSync(tempRoot, { recursive: true, force: true })
}
})
+166
View File
@@ -0,0 +1,166 @@
import type { LocalCatalogModel, LocalHardware, LocalModelsStatus, LocalRuntimeJob } from '@/types/hermes'
import { hermesApi, profileScoped } from './client'
// The desktop surface of the managed llama.cpp runtime: status/catalog
// reads, download/install/activate jobs, and server control.
export function getLocalModelsStatus(): Promise<LocalModelsStatus> {
return hermesApi<LocalModelsStatus>({
...profileScoped(),
path: '/api/local-models/status'
})
}
export function getLocalHardware(): Promise<LocalHardware> {
return hermesApi<LocalHardware>({
...profileScoped(),
path: '/api/local-models/hardware'
})
}
export function getLocalCatalog(): Promise<{ models: LocalCatalogModel[] }> {
return hermesApi<{ models: LocalCatalogModel[] }>({
...profileScoped(),
path: '/api/local-models/catalog'
})
}
export function installLocalRuntime(backend?: string): Promise<{ backend: string; job_id: string; tag: string }> {
return hermesApi<{ backend: string; job_id: string; tag: string }>({
...profileScoped(),
body: { backend: backend ?? null },
method: 'POST',
path: '/api/local-models/runtime/install'
})
}
export interface QuickstartResponse {
display_name: string
download_bytes: number
job_id: string
model_id: string
needs_download: boolean
needs_runtime: boolean
}
export function quickstartLocalModels(modelId?: string): Promise<QuickstartResponse> {
return hermesApi<QuickstartResponse>({
...profileScoped(),
body: { model_id: modelId ?? null },
method: 'POST',
path: '/api/local-models/quickstart'
})
}
export function downloadLocalModel(modelId: string): Promise<{ already_downloaded?: boolean; job_id: null | string }> {
return hermesApi<{ already_downloaded?: boolean; job_id: null | string }>({
...profileScoped(),
body: { model_id: modelId },
method: 'POST',
path: '/api/local-models/download'
})
}
export function deleteLocalModel(modelId: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
method: 'DELETE',
path: `/api/local-models/models/${encodeURIComponent(modelId)}`
})
}
export function getLocalRuntimeJob(jobId: string): Promise<LocalRuntimeJob> {
return hermesApi<LocalRuntimeJob>({
...profileScoped(),
path: `/api/local-models/jobs/${encodeURIComponent(jobId)}`
})
}
export function getLocalModelsJobs(): Promise<{ jobs: LocalRuntimeJob[] }> {
return hermesApi<{ jobs: LocalRuntimeJob[] }>({
...profileScoped(),
path: '/api/local-models/jobs'
})
}
export function activateLocalModel(modelId: string): Promise<{ job_id: string }> {
return hermesApi<{ job_id: string }>({
...profileScoped(),
body: { model_id: modelId },
method: 'POST',
path: '/api/local-models/activate'
})
}
export function ejectLocalModel(modelId: string): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
body: { model_id: modelId },
method: 'POST',
path: '/api/local-models/eject'
})
}
export function setLocalServer(action: 'start' | 'stop'): Promise<{ ok: boolean }> {
return hermesApi<{ ok: boolean }>({
...profileScoped(),
body: { action },
method: 'POST',
path: '/api/local-models/server'
})
}
// ── Hugging Face browser + sideload ─────────────────────────────
export interface HFSearchHit {
repo: string
downloads: number
likes: number
updated: string
gated: boolean
}
export interface HFFileGroup {
label: string
paths: string[]
total_bytes: number
fit: 'fits-gpu' | 'needs-ram' | 'too-big' | 'unknown'
}
export function searchHFModels(q: string, limit = 20): Promise<{ hits: HFSearchHit[] }> {
return hermesApi<{ hits: HFSearchHit[] }>({
...profileScoped(),
path: `/api/local-models/search?q=${encodeURIComponent(q)}&limit=${limit}`
})
}
export function listHFRepoFiles(repo: string): Promise<{ files: HFFileGroup[] }> {
return hermesApi<{ files: HFFileGroup[] }>({
...profileScoped(),
path: `/api/local-models/search/files?repo=${encodeURIComponent(repo)}`
})
}
export function downloadBrowsedModel(
repo: string,
paths: string[]
): Promise<{ already_downloaded?: boolean; job_id: null | string; model_id: string }> {
return hermesApi<{ already_downloaded?: boolean; job_id: null | string; model_id: string }>({
...profileScoped(),
body: { paths, repo },
method: 'POST',
path: '/api/local-models/download-browsed'
})
}
export function sideloadLocalModel(
path: string
): Promise<{ already_present?: boolean; model_id: string; ok: boolean }> {
return hermesApi<{ already_present?: boolean; model_id: string; ok: boolean }>({
...profileScoped(),
body: { path },
method: 'POST',
path: '/api/local-models/sideload'
})
}
+33 -8
View File
@@ -148,6 +148,11 @@ export interface SidebarSessionSlice {
/** Per-profile tokens and spend over every session, not just this window.
* Absent from the legacy per-slice endpoint, which has no aggregate. */
profiles_usage?: Record<string, { cost_usd: number; tokens: number }>
/** Profiles whose scan for THIS slice failed. Batched `/sidebar` stamps the
* same profile errors on every slice (one DB open). Legacy per-slice calls
* stamp only the slice that actually failed, so a cron I/O error cannot
* carry-forward recents. */
errors?: Array<{ profile: string; error: string }>
}
/** Which profiles filled their per-profile window in a returned page. The
@@ -216,16 +221,24 @@ async function listSidebarSessionsLegacy(req: SidebarSessionsRequest): Promise<S
})
])
const errors = [...(recents.errors ?? []), ...(cron.errors ?? []), ...(messaging.errors ?? [])]
const recentsErrors = recents.errors ?? []
const cronErrors = cron.errors ?? []
const messagingErrors = messaging.errors ?? []
return {
recents: {
profiles_truncated: profilesTruncatedFrom(recents.sessions, req.recentsLimit),
sessions: recents.sessions
sessions: recents.sessions,
...(recentsErrors.length ? { errors: recentsErrors } : {})
},
cron: { sessions: cron.sessions },
messaging: { sessions: messaging.sessions },
...(errors.length ? { errors } : {})
cron: {
sessions: cron.sessions,
...(cronErrors.length ? { errors: cronErrors } : {})
},
messaging: {
sessions: messaging.sessions,
...(messagingErrors.length ? { errors: messagingErrors } : {})
}
}
}
@@ -288,9 +301,21 @@ export async function listSidebarSessions(req: SidebarSessionsRequest): Promise<
}
return {
recents: { ...result.recents, sessions: stampActiveConnectionOwner(result.recents?.sessions ?? []) },
cron: { ...result.cron, sessions: stampActiveConnectionOwner(result.cron?.sessions ?? []) },
messaging: { ...result.messaging, sessions: stampActiveConnectionOwner(result.messaging?.sessions ?? []) },
recents: {
...result.recents,
sessions: stampActiveConnectionOwner(result.recents?.sessions ?? []),
...(result.errors?.length ? { errors: result.errors } : {})
},
cron: {
...result.cron,
sessions: stampActiveConnectionOwner(result.cron?.sessions ?? []),
...(result.errors?.length ? { errors: result.errors } : {})
},
messaging: {
...result.messaging,
sessions: stampActiveConnectionOwner(result.messaging?.sessions ?? []),
...(result.errors?.length ? { errors: result.errors } : {})
},
errors: result.errors
}
}
+21
View File
@@ -3,6 +3,7 @@ import type {
ActionStatusResponse,
AudioSpeakResponse,
AudioTranscriptionResponse,
AudioTtsLeaseResponse,
BackendUpdateCheckResponse,
CuratorStatusResponse,
DebugShareResponse,
@@ -196,6 +197,26 @@ export function speakText(text: string): Promise<AudioSpeakResponse> {
})
}
// Acquiring a lease pre-loads the configured TTS engine. For local engines
// that is a model load and, on a fresh install, a voice download — well past
// the default 15s Electron backend timeout.
export const AUDIO_TTS_LEASE_REQUEST_TIMEOUT_MS = 180_000
/**
* Tell the backend a speech-output toggle flipped so it can warm the TTS engine
* (`active: true`) or release it once no surface needs it (`active: false`).
* `lease` names the toggle — `desktop:read-aloud`, `desktop:conversation`.
*/
export function setTtsLease(lease: string, active: boolean): Promise<AudioTtsLeaseResponse> {
return hermesApi<AudioTtsLeaseResponse>({
...profileScoped(),
path: '/api/audio/tts-lease',
method: 'POST',
body: { active, lease },
timeoutMs: AUDIO_TTS_LEASE_REQUEST_TIMEOUT_MS
})
}
export function getElevenLabsVoices(profile?: null | string): Promise<ElevenLabsVoicesResponse> {
return hermesApi<ElevenLabsVoicesResponse>({
path: '/api/audio/elevenlabs/voices',
+42 -2
View File
@@ -143,19 +143,56 @@ const flatten = (nodes: readonly SubagentNode[]): SubagentNode[] =>
interface RootGroup {
id: string
delegationIndex: number
/** Short batch tag (`deleg_6a664903` → `6a66`) when the backend sent one. */
batchTag?: string
nodes: SubagentNode[]
taskCount: number
}
/** `deleg_6a664903` → `6a66`; mirrors tools.delegate_tool.format_batch_tag. */
export const batchTagOf = (delegationId: string | undefined): string | undefined => {
if (!delegationId) {
return undefined
}
const short = delegationId.split('_').at(-1)?.slice(0, 4)
return short || undefined
}
function groupDelegations(roots: readonly SubagentNode[]): RootGroup[] {
const groups: RootGroup[] = []
let n = 0
for (const node of roots) {
// Exact grouping when the backend tags workers with their batch id —
// concurrent or nested fan-outs of the same shape must not merge.
if (node.delegationId) {
const byId = groups.find(g => g.id === `delegation:${node.delegationId}`)
if (byId) {
byId.nodes.push(node)
continue
}
n += 1
groups.push({
id: `delegation:${node.delegationId}`,
delegationIndex: n,
batchTag: batchTagOf(node.delegationId),
nodes: [node],
taskCount: node.taskCount
})
continue
}
// Older backends (no delegation_id): heuristic grouping by shape + time.
const prev = groups.at(-1)
const prevTail = prev?.nodes.at(-1)
const closeInTime = prevTail ? Math.abs(node.startedAt - prevTail.startedAt) <= 5_000 : false
const sameShape = prev && node.taskCount > 1 && prev.taskCount === node.taskCount
const sameShape = prev && !prev.batchTag && node.taskCount > 1 && prev.taskCount === node.taskCount
const uniqueStep = prev ? !prev.nodes.some(item => item.taskIndex === node.taskIndex) : false
if (prev && sameShape && closeInTime && uniqueStep) {
@@ -248,7 +285,10 @@ function DelegationGroup({ group, nowMs }: { group: RootGroup; nowMs: number })
return (
<section className="grid min-w-0 gap-3">
<p className="text-[0.66rem] font-medium uppercase tracking-wider text-muted-foreground/70">
{group.delegationIndex > 0 ? t.agents.delegation(group.delegationIndex) : ''}{' '}
{group.delegationIndex > 0 ? t.agents.delegation(group.delegationIndex) : ''}
{group.batchTag ? (
<span className="ml-1 font-mono text-muted-foreground/60">[{group.batchTag}]</span>
) : null}{' '}
<span className="text-muted-foreground/50">·</span> {t.agents.workers(group.nodes.length)}
{activeWorkers > 0 ? <span className="text-primary/85"> · {t.agents.workersActive(activeWorkers)}</span> : null}
</p>
@@ -7,6 +7,7 @@ import {
isPendingDraftPersistCurrent,
type PendingDraftPersist,
pickPlaceholder,
shouldDisableComposerInput,
slashArgStage,
slashChipKindForItem,
slashCommandToken,
@@ -16,6 +17,26 @@ import {
const item = (group: string): Unstable_TriggerItem =>
({ id: 'x', type: 'slash', label: 'x', metadata: { group } }) as unknown as Unstable_TriggerItem
describe('shouldDisableComposerInput', () => {
it.each(['idle', 'connecting', 'closed', 'error'] as const)(
'keeps the draft editable while the gateway is %s',
gatewayState => {
expect(shouldDisableComposerInput(true, gatewayState)).toBe(false)
}
)
it('fails closed when connection atoms disagree about an open gateway', () => {
expect(shouldDisableComposerInput(true, 'open')).toBe(true)
})
it.each(['idle', 'connecting', 'open', 'closed', 'error'] as const)(
'never disables an otherwise enabled composer while the gateway is %s',
gatewayState => {
expect(shouldDisableComposerInput(false, gatewayState)).toBe(false)
}
)
})
describe('slashArgStage', () => {
it('is true only once the query is past the command name', () => {
expect(slashArgStage('personality')).toBe(false)
@@ -1,4 +1,5 @@
import type { Unstable_TriggerItem } from '@assistant-ui/core'
import type { ConnectionState } from '@hermes/shared'
import type { SlashChipKind } from '@/components/assistant-ui/directive-text'
import type { ComposerAttachment } from '@/store/composer'
@@ -52,6 +53,18 @@ export const COMPOSER_FADE_BACKGROUND =
// unmount/pagehide flushes bypass it.
export const DRAFT_PERSIST_DEBOUNCE_MS = 400
/**
* Keep a reconnecting draft editable so transient gateway dials cannot blur
* the editor and discard the user's caret. Submission still reads the
* independent `disabled` prop, so non-open states cannot send.
*
* An `open` state paired with `disabled=true` is a transient disagreement
* between the connection atoms; fail closed until they converge.
*/
export function shouldDisableComposerInput(disabled: boolean, gatewayState: ConnectionState): boolean {
return disabled && gatewayState === 'open'
}
export const pickPlaceholder = (pool: readonly string[]) => pool[Math.floor(Math.random() * pool.length)]
/** Completion items can carry an `action` (set in use-slash-completions) that
@@ -5,6 +5,7 @@ import { useI18n } from '@/i18n'
import { chatMessageText, collectUnspokenTurnSpeech } from '@/lib/chat-messages'
import { triggerHaptic } from '@/lib/haptics'
import { markAssistantIdSpoken, resolveSpokenReply } from '@/lib/spoken-reply'
import { CONVERSATION_LEASE, READ_ALOUD_LEASE, syncTtsLease } from '@/lib/tts-lease'
import { clearWakeIndicator, syncWakeIndicatorWithVoice } from '@/lib/wake-indicator'
import { $voiceConversationStartRequest, takeVoiceConversationStart } from '@/store/composer'
import { resetBrowseState } from '@/store/composer-input-history'
@@ -265,6 +266,26 @@ export function useComposerVoice({
useEffect(() => resumeWakeIfPaused, [resumeWakeIfPaused])
// Speech-output toggles are TTS warm-up / release signals. Entering a voice
// conversation acquires this window's lease (pre-loads the engine so the
// first spoken reply doesn't start with dead air); ending it releases the
// lease, and the backend unloads resident local models once no surface holds
// one. Fire-and-forget — the toggle never waits on or fails from this.
useEffect(() => {
void syncTtsLease(CONVERSATION_LEASE, voiceConversationActive)
}, [voiceConversationActive])
useEffect(() => () => void syncTtsLease(CONVERSATION_LEASE, false), [])
// "Read replies aloud" is the same signal, held for as long as the toggle is
// on (it mirrors voice.auto_tts, so this also warms at startup when the
// preference is already set).
const autoSpeakReplies = useStore($autoSpeakReplies)
useEffect(() => {
void syncTtsLease(READ_ALOUD_LEASE, autoSpeakReplies)
}, [autoSpeakReplies])
// Explicit start/end for the on-screen conversation controls (the hotkey uses
// the gated toggle above).
const startConversation = useCallback(() => setVoiceConversationActive(true), [])
+3 -2
View File
@@ -35,6 +35,7 @@ import {
COMPOSER_FADE_BACKGROUND,
implicitSlashAcceptIndex,
type QueueEditState,
shouldDisableComposerInput,
slashArgStage
} from './composer-utils'
import { ContextMenu } from './context-menu'
@@ -220,8 +221,8 @@ export function ChatBar({
const { t } = useI18n()
const gatewayState = useStore($gatewayState)
const reconnecting = gatewayState === 'closed' || gatewayState === 'error'
const inputDisabled = disabled && !reconnecting
const reconnecting = gatewayState !== 'open'
const inputDisabled = shouldDisableComposerInput(disabled, gatewayState)
// The draft engine — detached source of truth (DOM + draftRef + edge
// selectors); typing never re-renders the chrome. ChatBar owns `queueEditRef`
@@ -1,13 +0,0 @@
import { readFileSync } from 'node:fs'
import { resolve } from 'node:path'
import { describe, expect, it } from 'vitest'
const source = readFileSync(resolve(process.cwd(), 'src/app/chat/session-tile.tsx'), 'utf8')
describe('SessionTilePane owner-scoped listing', () => {
it('resolves a newly active tile on its persisted owner route', () => {
expect(source).toContain('void resolveStoredSession(storedSessionId, ownerRoute)')
expect(source).not.toMatch(/void resolveStoredSession\(storedSessionId\)\s*\n/)
})
})
@@ -0,0 +1,71 @@
import { beforeEach, describe, expect, it } from 'vitest'
import { _resetSessionOwnerHintsForTests, setSessionOwnerHint } from '@/store/session'
import type { SessionTile } from '@/store/session-states'
import type { SessionInfo } from '@/types/hermes'
import { tileOwnerRoute } from './session-tile-owner'
const row = (over: Partial<SessionInfo>): SessionInfo => over as SessionInfo
const tile = (over: Partial<SessionTile> & Pick<SessionTile, 'storedSessionId'>): SessionTile => over as SessionTile
describe('tileOwnerRoute', () => {
beforeEach(() => {
_resetSessionOwnerHintsForTests()
})
it('prefers the tile own explicit route', () => {
const route = tileOwnerRoute(
[tile({ ownerRoute: { connectionId: 'pandora', profile: 'work' }, storedSessionId: 's1' })],
[row({ connection_id: 'other-box', id: 's1', profile: 'default' })],
's1'
)
expect(route).toEqual({ connectionId: 'pandora', profile: 'work' })
})
it('falls back to the session row owner when the tile carries no route', () => {
// How a branch child is opened: openSessionTile with no workspaceScope, so
// the tile route alone leaves the owner undefined and every RPC drops to
// the ambient socket.
const route = tileOwnerRoute(
[tile({ storedSessionId: 's1' })],
[row({ connection_id: 'rigremote', id: 's1', profile: 'default' })],
's1'
)
expect(route).toEqual({ connectionId: 'rigremote', profile: 'default' })
})
it('falls back to the owner hint when neither tile nor row is tagged', () => {
setSessionOwnerHint('s1', { connectionId: 'pandora', profile: 'work' })
expect(tileOwnerRoute([tile({ storedSessionId: 's1' })], [], 's1')).toMatchObject({ connectionId: 'pandora' })
})
it('carries a targetProfile through, and omits it when absent', () => {
const routed = tileOwnerRoute(
[tile({ ownerRoute: { connectionId: 'pandora', profile: 'work', targetProfile: 'ceo' }, storedSessionId: 's1' })],
[],
's1'
)
expect(routed).toEqual({ connectionId: 'pandora', profile: 'work', targetProfile: 'ceo' })
expect(
tileOwnerRoute([tile({ ownerRoute: { connectionId: 'p', profile: 'w' }, storedSessionId: 's1' })], [], 's1')
).not.toHaveProperty('targetProfile')
})
it('narrows a bare profile owner away', () => {
// knownSessionOwner returns a bare profile string for a row that names a
// profile but no connection. It carries no backend identity, so handing it
// on as a route would resolve against whichever connection is active.
expect(tileOwnerRoute([], [row({ id: 's1', profile: 'work' })], 's1')).toBeUndefined()
})
it('is undefined for an untagged session, preserving ambient routing', () => {
expect(tileOwnerRoute([], [row({ id: 's1' })], 's1')).toBeUndefined()
expect(tileOwnerRoute([], [], 'missing')).toBeUndefined()
})
})
@@ -0,0 +1,37 @@
import { knownSessionOwner } from '@/store/session'
import type { SessionOwnerRoute, SessionOwnerScope } from '@/store/session-request-router'
import type { SessionTile } from '@/store/session-states'
import type { SessionInfo } from '@/types/hermes'
/**
* The owner a session tile routes its own RPCs through — the tile's explicit
* route first, then the session row's `(connection, profile)` tag, with
* `knownSessionOwner` folding in the owner hint.
*
* A tile opened without an explicit route — a branch child, which
* `openSessionTile` creates with no `workspaceScope` — has no tile route, so
* the row/hint rung is the only thing keeping its model and composer RPCs on
* the backend that owns the session instead of the ambient one.
*
* A bare profile string carries no connection and is not a usable route:
* handing it to `requestForSessionProfile` would resolve it against whichever
* connection is active, which is the bug this ladder exists to avoid.
*/
export function tileOwnerRoute(
tiles: readonly SessionTile[],
rows: readonly SessionInfo[],
storedSessionId: string
): SessionOwnerRoute | undefined {
const owner: SessionOwnerScope =
tiles.find(tile => tile.storedSessionId === storedSessionId)?.ownerRoute ?? knownSessionOwner(rows, storedSessionId)
if (!owner || typeof owner !== 'object' || !owner.connectionId) {
return undefined
}
return {
connectionId: owner.connectionId,
profile: owner.profile,
...(owner.targetProfile ? { targetProfile: owner.targetProfile } : {})
}
}
+45 -2
View File
@@ -1,6 +1,9 @@
import { describe, expect, it } from 'vitest'
import { afterEach, describe, expect, it, vi } from 'vitest'
import { sessionTileResumeFailure } from './session-tile'
import { $gatewayState, $sessions, setSessions } from '@/store/session'
import { $sessionTiles } from '@/store/session-states'
import { sessionTileResumeFailure, startUnrestoredTileTitleBackfill } from './session-tile'
describe('sessionTileResumeFailure', () => {
it('keeps a confirmed durable session retryable instead of repeating a stale 404', () => {
@@ -17,3 +20,43 @@ describe('sessionTileResumeFailure', () => {
expect(sessionTileResumeFailure('session not found', true, false)).toBeUndefined()
})
})
describe('startUnrestoredTileTitleBackfill (#94167)', () => {
afterEach(() => {
$gatewayState.set('idle')
$sessionTiles.set([])
setSessions([])
})
it('backfills unlisted unrestored tiles by id via their ownerRoute once the gateway opens', async () => {
const ownerRoute = { connectionId: 'conn-a', profile: 'writer' }
setSessions([{ id: 'listed', title: 'Already listed' } as never])
$sessionTiles.set([
{ ownerRoute, storedSessionId: 'old-chat' },
{ storedSessionId: 'listed' },
{ runtimeId: 'rt-live', storedSessionId: 'live' },
{ storedSessionId: 'bot', workspaceTabTitle: 'Bot Chat' }
])
const lookup = vi.fn(async (id: string) => {
const row = { id, title: 'Quarterly review' } as never
setSessions(prev => [row, ...prev])
return row
})
const stop = startUnrestoredTileTitleBackfill(lookup as never)
expect(lookup).not.toHaveBeenCalled()
$gatewayState.set('open')
await vi.waitFor(() => expect(lookup).toHaveBeenCalledTimes(1))
expect(lookup).toHaveBeenCalledWith('old-chat', ownerRoute)
expect($sessions.get().find(row => row.id === 'old-chat')?.title).toBe('Quarterly review')
// One-shot: a later reconnect does not re-probe.
$gatewayState.set('idle')
$gatewayState.set('open')
expect(lookup).toHaveBeenCalledTimes(1)
stop()
})
})
+61 -8
View File
@@ -28,6 +28,7 @@ import { formatRefValue } from '@/components/assistant-ui/directive-text'
import { CenteredThreadSpinner } from '@/components/assistant-ui/thread/status'
import { findGroupOfPane } from '@/components/pane-shell/tree/model'
import { $layoutTree, closeTreePane, moveTreePane, setTreeGroupTabStrip } from '@/components/pane-shell/tree/store'
import { $workspaceOwnerLabels, workspaceOwnerTitle } from '@/components/pane-shell/workspace-scope'
import { Button } from '@/components/ui/button'
import { ConfirmDialog } from '@/components/ui/confirm-dialog'
import { transcribeAudio } from '@/hermes'
@@ -41,7 +42,9 @@ import { $activeGatewayProfile } from '@/store/profile'
import { $projectTree } from '@/store/projects'
import { sessionAwaitingInput } from '@/store/prompts'
import {
$cronSessions,
$gatewayState,
$messagingSessions,
$selectedStoredSessionId,
$sessions,
sessionMatchesStoredId,
@@ -55,8 +58,7 @@ import {
closeSessionTile,
patchSessionTile,
type SessionTile,
sessionTileDelegate,
sessionTileOwnerRoute
sessionTileDelegate
} from '@/store/session-states'
import type { SessionInfo } from '@/types/hermes'
@@ -68,6 +70,7 @@ import { SessionDraftTitle } from './session-draft-title'
import { startSessionDrag } from './session-drag'
import { SessionStatusDot } from './session-status-dot'
import { useSessionTileActions } from './session-tile-actions'
import { tileOwnerRoute } from './session-tile-owner'
import { type SessionView, SessionViewProvider } from './session-view'
import { SessionContextMenu } from './sidebar/session-actions-menu'
import { lastVisibleMessageIsUser } from './thread-loading'
@@ -157,7 +160,21 @@ function TileChat({
}) {
const { gateway, requestGateway } = useGatewayRequest()
const queryClient = useQueryClient()
const ownerRoute = sessionTileOwnerRoute(storedSessionId)
// Owner ladder, same as useSessionTileActions (session-tile-actions.ts:99-103).
// Recomputed when the tile store or any owner-bearing session list changes,
// NOT on every render: this component re-renders per streamed token, and the
// lookup spreads three arrays before scanning them.
const tiles = useStore($sessionTiles)
const sessionRows = useStore($sessions)
const cronRows = useStore($cronSessions)
const messagingRows = useStore($messagingSessions)
const ownerRoute = useMemo(() => {
const rows = cronRows.length || messagingRows.length ? [...sessionRows, ...cronRows, ...messagingRows] : sessionRows
return tileOwnerRoute(tiles, rows, storedSessionId)
}, [cronRows, messagingRows, sessionRows, storedSessionId, tiles])
const requestTileGateway = useCallback(
<T,>(method: string, params?: Record<string, unknown>, timeoutMs?: number, signal?: AbortSignal): Promise<T> =>
@@ -461,6 +478,33 @@ export function tileStoredRow(storedSessionId: string): SessionInfo | undefined
)
}
/** One-shot by-id title fill for restored tiles that never mount (#94167).
* A restored background tab has no runtimeId and does not mount its pane, so
* the resolution effect above never runs; when its row is outside the recents
* page and project tree, `tileTitle()` reads "New session" until first click.
* `resolveStoredSession` upserts the row into `$sessions`, which the tab strip
* already watches — nothing is persisted. Runs once the gateway can answer. */
export function startUnrestoredTileTitleBackfill(lookup = resolveStoredSession): () => void {
const run = () => {
if ($gatewayState.get() !== 'open') {
return
}
off()
for (const tile of $sessionTiles.get()) {
if (!tile.runtimeId && !tile.workspaceTabTitle && !tileStoredRow(tile.storedSessionId)) {
void lookup(tile.storedSessionId, tile.ownerRoute).catch(() => undefined)
}
}
}
const off = $gatewayState.listen(run)
run()
return off
}
/** The tab's REGISTERED name. Deliberately the bare placeholder for a draft
* rather than its live composer title (`tabTitle` renders that): re-registering
* per keystroke would re-render the strip, and holding the draft's text here
@@ -473,14 +517,23 @@ function tileTitle(storedSessionId: string): string {
return stored ? sessionTitle(stored) : explicit || NEW_SESSION_TITLE
}
/** The tab's CAPTION: a bot chat's owner name over the canonical stored title
* (#99152). The menu keeps `tileTitle` — rename/delete show the real row. */
function tileCaption(storedSessionId: string): string {
return workspaceOwnerTitle(
tileTitle(storedSessionId),
$sessionTiles.get().find(tile => tile.storedSessionId === storedSessionId)
)
}
/** The `@session` link payload for a tile tab drag — id + owning profile + title.
* Resolved at drag time, so an unsent tab drags under its draft name. */
function tileDragPayload(storedSessionId: string): SessionDragPayload {
const stored = tileStoredRow(storedSessionId)
const explicit = $sessionTiles.get().find(tile => tile.storedSessionId === storedSessionId)?.workspaceTabTitle
const title = stored ? sessionTitle(stored) : explicit || draftTitleFor(storedSessionId) || NEW_SESSION_TITLE
const tile = $sessionTiles.get().find(candidate => candidate.storedSessionId === storedSessionId)
const title = stored ? sessionTitle(stored) : tile?.workspaceTabTitle || draftTitleFor(storedSessionId) || NEW_SESSION_TITLE
return { id: storedSessionId, profile: stored?.profile ?? '', title }
return { id: storedSessionId, profile: stored?.profile ?? '', title: workspaceOwnerTitle(title, tile) }
}
// ---------------------------------------------------------------------------
@@ -667,14 +720,14 @@ export const watchSessionTiles = paneMirror<SessionTile>({
// $projectTree: a tile whose session is older than the recents page resolves
// its title through the tree, which loads after the tiles register. (The tab's
// status dot subscribes to color/state itself, so it needs no `also` entry.)
also: [$sessions, $projectTree],
also: [$sessions, $projectTree, $workspaceOwnerLabels],
key: t => t.storedSessionId,
prefix: 'session-tile',
dir: t => t.dir,
anchor: t => t.anchor,
before: t => t.before,
minWidth: '20rem',
title: tileTitle,
title: tileCaption,
// The tab's status dot — the SAME primitive the sidebar row renders, keyed by
// the stored id, so a session's status/color can never disagree between the
// two surfaces. Self-subscribing (live state + resolved color), so the strip
+12 -1
View File
@@ -34,6 +34,7 @@ import {
toggleTargetZoneTabStrip,
watchContributedPanes
} from '@/components/pane-shell/tree/store'
import { $workspaceOwnerLabels, workspaceOwnerTitle } from '@/components/pane-shell/workspace-scope'
import { SidebarProvider } from '@/components/ui/sidebar'
import { discoverBundledPlugins } from '@/contrib/plugins'
import { Slot } from '@/contrib/react/slot'
@@ -70,6 +71,7 @@ import {
} from '@/store/review'
import { $currentCwd, $selectedStoredSessionId, $sessions, $yoloActive, sessionMatchesStoredId } from '@/store/session'
import { watchSessionPins } from '@/store/session-pin-sync'
import { $botChatScopes } from '@/store/session-states'
import { watchUnreadWriteGuard } from '@/store/session-unread-remote'
import { $statusbarVisible } from '@/store/statusbar-prefs'
import { isBrowserWindow, isHudWindow } from '@/store/windows'
@@ -82,6 +84,7 @@ import { startSessionDrag } from '../chat/session-drag'
import {
SessionTileCloseConfirm,
stackSessionTilesIntoMain,
startUnrestoredTileTitleBackfill,
watchSessionTiles,
WorkspaceTabMenu
} from '../chat/session-tile'
@@ -457,6 +460,7 @@ watchContributedPanes()
// into the transparent overlay).
if (!isBrowserWindow() && !isHudWindow()) {
watchSessionTiles()
startUnrestoredTileTitleBackfill()
watchRouteTiles()
watchPreviewTiles()
}
@@ -490,7 +494,12 @@ const syncWorkspaceTitle = () => {
area: 'panes',
// The placeholder, not the draft's live name — `tabTitle` below renders
// that. Keeping it here would re-register the pane on every keystroke.
title: stored ? storedSessionTitle(stored) : NEW_SESSION_TITLE,
// A bot chat reads as its BOT: every canonical Bot Chat is stored under
// the same name, which told two open bots apart by nothing (#99152).
title: workspaceOwnerTitle(
stored ? storedSessionTitle(stored) : NEW_SESSION_TITLE,
selected ? $botChatScopes.get()[selected] : undefined
),
data: {
// The tab's status dot — the SAME primitive the sidebar row and session
// tiles render, so the main tab never disagrees with its sidebar row. A
@@ -515,6 +524,8 @@ const syncWorkspaceTitle = () => {
$selectedStoredSessionId.listen(syncWorkspaceTitle)
$sessions.listen(syncWorkspaceTitle)
$botChatScopes.listen(syncWorkspaceTitle)
$workspaceOwnerLabels.listen(syncWorkspaceTitle)
$workspaceIsPage.listen(syncWorkspaceTitle)
// Layout reset collapses every session tile into main as a tab (after the
@@ -8,15 +8,18 @@ import {
$activeSessionId,
$selectedStoredSessionId,
setBusy,
setCronSessions,
setMessagingSessions,
setSessionOwnerHint,
setSessions
} from '@/store/session'
import {
$attentionSessionIds,
$sessionTiles,
$stalledSessionIds,
$workingSessionIds,
clearAllSessionStates,
publishSessionState,
SESSION_WATCHDOG_TIMEOUT_MS
} from '@/store/session-states'
@@ -38,7 +41,13 @@ vi.mock('@/hermes', async importOriginal => ({
getLatestSessionMessages: vi.fn()
}))
vi.mock('@/store/projects', async importOriginal => ({
...(await importOriginal()),
refreshProjectTree: vi.fn(async () => undefined)
}))
const { getLatestSessionMessages } = await import('@/hermes')
const { refreshProjectTree } = await import('@/store/projects')
const ACTIVE_RUNTIME_ID = 'runtime-active'
const ACTIVE_STORED_ID = 'stored-active'
@@ -91,11 +100,13 @@ function useSyncHarness({
activeIsMessaging = false,
activeSessionId,
activeStoredSessionId,
gatewayState = 'open',
refreshActiveTranscript
}: {
activeIsMessaging?: boolean
activeSessionId: string | null
activeStoredSessionId: string | null
gatewayState?: string
refreshActiveTranscript: () => Promise<void>
}) {
const updateSessionState: Parameters<typeof useBackgroundSync>[0]['updateSessionState'] = vi.fn(
@@ -113,7 +124,7 @@ function useSyncHarness({
activeSessionId,
activeStoredSessionId,
freshDraftReady: false,
gatewayState: 'open',
gatewayState,
refreshActiveTranscript,
refreshCronJobs: vi.fn(),
refreshCurrentModel: vi.fn(),
@@ -125,17 +136,23 @@ function useSyncHarness({
})
}
function renderSync(
refreshActiveTranscript: () => Promise<void>,
options: { activeIsMessaging?: boolean; activeSessionId?: null | string; activeStoredSessionId?: null | string } = {}
) {
return renderHook(() =>
useSyncHarness({
activeSessionId: ACTIVE_RUNTIME_ID,
activeStoredSessionId: ACTIVE_STORED_ID,
refreshActiveTranscript,
...options
})
type SyncOptions = {
activeIsMessaging?: boolean
activeSessionId?: null | string
activeStoredSessionId?: null | string
gatewayState?: string
}
function renderSync(refreshActiveTranscript: () => Promise<void>, options: SyncOptions = {}) {
return renderHook(
(props: SyncOptions) =>
useSyncHarness({
activeSessionId: ACTIVE_RUNTIME_ID,
activeStoredSessionId: ACTIVE_STORED_ID,
refreshActiveTranscript,
...props
}),
{ initialProps: options }
)
}
@@ -153,11 +170,13 @@ afterEach(() => {
$activeSessionId.set(null)
$selectedStoredSessionId.set(null)
setSessions([])
setCronSessions([])
setMessagingSessions([])
setBusy(false)
vi.clearAllMocks()
vi.restoreAllMocks()
clearAllSessionStates()
$sessionTiles.set([])
resetTypingActivityTracking()
})
@@ -228,7 +247,6 @@ describe('active transcript refresh', () => {
const signatureRef = { current: new Map<string, string>() }
const requestSequenceRef = { current: 0 }
const busyRef = { current: false }
vi.mocked(getLatestSessionMessages).mockImplementation(async (storedId: string) => {
if (storedId === TILE_STORED_ID) {
@@ -250,7 +268,6 @@ describe('active transcript refresh', () => {
await act(async () => {
await reconcileTileTranscriptsForTest({
tiles: [{ storedSessionId: TILE_STORED_ID, runtimeId: TILE_RUNTIME_ID }],
busyRef,
requestSequenceRef,
signatureRef,
updateSessionState
@@ -259,7 +276,128 @@ describe('active transcript refresh', () => {
// Behavior assertions:
expect(updaterCallCount).toBeGreaterThan(0)
expect(getLatestSessionMessages).toHaveBeenCalledWith(TILE_STORED_ID)
expect(getLatestSessionMessages).toHaveBeenCalledWith(TILE_STORED_ID, undefined)
})
it('reconciles an idle tile while the main pane is busy', async () => {
const runtimeId = 'runtime-idle-tile'
const storedId = 'stored-idle-tile'
const idleState = createClientSessionState(storedId)
setBusy(true)
publishSessionState(runtimeId, idleState)
vi.mocked(getLatestSessionMessages).mockResolvedValue(transcript('idle tile update', storedId) as never)
const updateSessionState = vi.fn((sessionId: string, updater: (state: typeof idleState) => typeof idleState) => {
expect(sessionId).toBe(runtimeId)
return updater(idleState)
})
await reconcileTileTranscriptsForTest({
tiles: [{ runtimeId, storedSessionId: storedId }],
requestSequenceRef: { current: 0 },
signatureRef: { current: new Map() },
updateSessionState
})
expect(getLatestSessionMessages).toHaveBeenCalledWith(storedId, undefined)
expect(updateSessionState).toHaveBeenCalledTimes(1)
})
it('does not reconcile a busy tile when the main pane is idle', async () => {
const runtimeId = 'runtime-busy-tile'
const storedId = 'stored-busy-tile'
const liveState = createClientSessionState(storedId)
liveState.busy = true
liveState.messages = [
{
id: 'live-assistant',
parts: [{ text: 'streaming answer', type: 'text' }],
pending: true,
role: 'assistant'
}
]
publishSessionState(runtimeId, liveState)
vi.mocked(getLatestSessionMessages).mockResolvedValue({ messages: [], session_id: storedId } as never)
const updateSessionState = vi.fn()
await reconcileTileTranscriptsForTest({
tiles: [{ runtimeId, storedSessionId: storedId }],
requestSequenceRef: { current: 0 },
signatureRef: { current: new Map() },
updateSessionState
})
expect(getLatestSessionMessages).not.toHaveBeenCalled()
expect(updateSessionState).not.toHaveBeenCalled()
})
it('discards a tile snapshot when the tile closes during the read', async () => {
const runtimeId = 'runtime-closing-tile'
const storedId = 'stored-closing-tile'
let resolveRead: (value: unknown) => void = () => undefined
$sessionTiles.set([{ runtimeId, storedSessionId: storedId }])
publishSessionState(runtimeId, createClientSessionState(storedId))
vi.mocked(getLatestSessionMessages).mockReturnValueOnce(
new Promise(resolve => {
resolveRead = resolve
}) as never
)
const updateSessionState = vi.fn()
const reconcile = reconcileTileTranscriptsForTest({
requestSequenceRef: { current: 0 },
signatureRef: { current: new Map() },
updateSessionState
})
$sessionTiles.set([])
resolveRead(transcript('stale tile answer', storedId))
await reconcile
expect(updateSessionState).not.toHaveBeenCalled()
})
it('isolates tile transcript reads by connection and profile while preserving the legacy local path', async () => {
vi.mocked(getLatestSessionMessages).mockImplementation(async storedId => transcript(storedId, storedId) as never)
const updateSessionState: Parameters<typeof reconcileTileTranscriptsForTest>[0]['updateSessionState'] = vi.fn(
(_sessionId, updater) => updater({} as Parameters<typeof updater>[0])
)
await reconcileTileTranscriptsForTest({
tiles: [
{
ownerRoute: { connectionId: 'connection-a', mode: 'remote', profile: 'shared-profile', targetProfile: 'target-a' },
runtimeId: 'runtime-a',
storedSessionId: 'stored-a'
},
{
ownerRoute: { connectionId: 'connection-b', mode: 'remote', profile: 'shared-profile' },
runtimeId: 'runtime-b',
storedSessionId: 'stored-b'
},
{ runtimeId: 'runtime-local', storedSessionId: 'stored-local' }
],
requestSequenceRef: { current: 0 },
signatureRef: { current: new Map<string, string>() },
updateSessionState
})
expect(getLatestSessionMessages).toHaveBeenCalledWith('stored-a', { connectionId: 'connection-a', profile: 'target-a' })
expect(getLatestSessionMessages).toHaveBeenCalledWith('stored-b', {
connectionId: 'connection-b',
profile: 'shared-profile'
})
expect(getLatestSessionMessages).toHaveBeenCalledWith('stored-local', undefined)
expect(updateSessionState).toHaveBeenCalledWith('runtime-a', expect.any(Function), 'stored-a')
expect(updateSessionState).toHaveBeenCalledWith('runtime-b', expect.any(Function), 'stored-b')
expect(updateSessionState).toHaveBeenCalledWith('runtime-local', expect.any(Function), 'stored-local')
})
it('skips the tile fetch entirely when nothing changed (signature-gated)', async () => {
@@ -287,13 +425,11 @@ describe('active transcript refresh', () => {
signatureRef.current.set(`tile:${TILE_STORED_ID}`, preSignature)
const updateSessionState = vi.fn()
const busyRef = { current: false }
const requestSequenceRef = { current: 0 }
await act(async () => {
await reconcileTileTranscriptsForTest({
tiles: [{ storedSessionId: TILE_STORED_ID, runtimeId: TILE_RUNTIME_ID }],
busyRef,
requestSequenceRef,
signatureRef,
updateSessionState
@@ -343,14 +479,15 @@ describe('active transcript refresh', () => {
const refresh = vi.fn(async () => undefined)
renderSync(refresh)
expect(refresh).not.toHaveBeenCalled()
// Exactly the one connect-time pull (#94779) — no timer after it.
expect(refresh).toHaveBeenCalledTimes(1)
await act(async () => {
vi.advanceTimersByTime(60_000)
await Promise.resolve()
})
expect(refresh).not.toHaveBeenCalled()
expect(refresh).toHaveBeenCalledTimes(1)
})
it('retains the existing periodic backstop for messaging sessions', async () => {
@@ -373,11 +510,12 @@ describe('active transcript refresh', () => {
it('only defers an external tick while busy, then refreshes once after idle', async () => {
$changeEventsAvailable.set(true)
setBusy(true)
const refresh = vi.fn(async () => undefined)
renderSync(refresh)
refresh.mockClear() // drop the connect-time pull; this test is about busy transitions
act(() => setBusy(true))
act(() => setBusy(false))
expect(refresh).not.toHaveBeenCalled()
act(() => setBusy(true))
@@ -392,12 +530,31 @@ describe('active transcript refresh', () => {
await waitFor(() => expect(refresh).toHaveBeenCalledTimes(1))
})
it('pulls the open transcript once per (re)connect, not on session switches (#94779)', () => {
$changeEventsAvailable.set(true)
const refresh = vi.fn(async () => undefined)
const { rerender } = renderSync(refresh, { gatewayState: 'connecting' })
expect(refresh).not.toHaveBeenCalled()
rerender({ gatewayState: 'open' })
expect(refresh).toHaveBeenCalledTimes(1)
rerender({ activeSessionId: 'runtime-other', activeStoredSessionId: 'stored-other', gatewayState: 'open' })
expect(refresh).toHaveBeenCalledTimes(1)
rerender({ activeSessionId: 'runtime-other', activeStoredSessionId: 'stored-other', gatewayState: 'closed' })
rerender({ activeSessionId: 'runtime-other', activeStoredSessionId: 'stored-other', gatewayState: 'open' })
expect(refresh).toHaveBeenCalledTimes(2)
})
it('coalesces a burst of global session-change ticks', async () => {
vi.useFakeTimers()
$changeEventsAvailable.set(true)
const refresh = vi.fn(async () => undefined)
renderSync(refresh)
refresh.mockClear() // drop the connect-time pull; this test is about tick coalescing
act(() => {
for (let index = 0; index < 20; index += 1) {
@@ -413,6 +570,16 @@ describe('active transcript refresh', () => {
expect(refresh).toHaveBeenCalledTimes(1)
})
it('refreshes the project tree on a sessions.changed tick, alongside the sessions list (#100354)', async () => {
$changeEventsAvailable.set(true)
renderSync(vi.fn(async () => undefined))
act(() => notifySessionsChanged())
await waitFor(() => expect(refreshProjectTree).toHaveBeenCalledTimes(1))
})
})
describe('reconcileActiveTranscript', () => {
@@ -435,6 +602,19 @@ describe('reconcileActiveTranscript', () => {
})
})
it('resolves and hydrates a cron session from the cron sessions store', async () => {
setCronSessions([{ id: ACTIVE_STORED_ID, profile: 'cron-profile', source: 'cron' } as never])
const fixture = makeRefresh(resolveActiveTranscriptSession)
vi.mocked(getLatestSessionMessages).mockResolvedValue(transcript('cron progress') as never)
await fixture.refresh()
expect(getLatestSessionMessages).toHaveBeenCalledWith(ACTIVE_STORED_ID, 'cron-profile')
expect(fixture.states.get(ACTIVE_RUNTIME_ID)?.messages.at(-1)?.parts[0]).toMatchObject({
text: 'cron progress'
})
})
it('fails closed when a hidden session id has multiple owner hints', async () => {
const ambiguousStoredSessionId = 'ambiguous-hidden-chat'
setSessionOwnerHint(ambiguousStoredSessionId, {
@@ -9,14 +9,14 @@ import { sessionMessagesSignature } from '@/lib/session-signatures'
import { $changeEventsAvailable, $cronChangeTick, $sessionsChangeTick } from '@/store/live-sync'
import { $onBattery, batteryPollInterval } from '@/store/power'
import { refreshActiveProfile } from '@/store/profile'
import { refreshProjectTree } from '@/store/projects'
import {
$activeSessionId,
$busy,
$currentCwd,
$messagingSessions,
$selectedStoredSessionId,
$sessions,
getSessionOwnerHint,
ownerLookupSessionRows,
sessionMatchesStoredId,
setCurrentCwd
} from '@/store/session'
@@ -39,9 +39,7 @@ interface ActiveTranscriptSession {
/** Resolve an active transcript from visible rows or its unique hidden owner. */
export function resolveActiveTranscriptSession(storedSessionId: string): ActiveTranscriptSession | undefined {
const visible =
$sessions.get().find(session => sessionMatchesStoredId(session, storedSessionId)) ??
$messagingSessions.get().find(session => sessionMatchesStoredId(session, storedSessionId))
const visible = ownerLookupSessionRows().find(session => sessionMatchesStoredId(session, storedSessionId))
if (visible) {
return { profile: visible.profile }
@@ -66,6 +64,22 @@ export interface ActiveTranscriptRefreshDeps {
) => ClientSessionState
}
function tileRuntimeOwnsLiveState(runtimeId: string): boolean {
const state = $sessionStates.get()[runtimeId]
return Boolean(state && (state.busy || state.awaitingResponse || state.needsInput || state.turnLive))
}
type TileTranscriptTarget = { ownerRoute?: SessionProfileRoute; storedSessionId: string; runtimeId?: string }
/** Signature key per tile — carries the owner route so two connections/profiles
* sharing a stored id (or a tile re-homed to another owner) never alias. */
function tileTranscriptSignatureKey(tile: TileTranscriptTarget): string {
const route = tile.ownerRoute
return `tile:${route ? `${route.connectionId}:${route.targetProfile ?? route.profile}:` : ''}${tile.storedSessionId}`
}
/**
* Reconcile the persisted transcripts of every open WORKSPACE TILE (#93942
* slice 1). Bot canonical chats live here — never in $sessions /
@@ -84,15 +98,13 @@ export interface ActiveTranscriptRefreshDeps {
*/
export async function reconcileTileTranscripts({
requestSequenceRef,
busyRef,
signatureRef,
updateSessionState,
tiles: tilesOverride
}: {
busyRef: MutableRefObject<boolean>
requestSequenceRef: MutableRefObject<number>
signatureRef: MutableRefObject<Map<string, string>>
tiles?: Array<{ storedSessionId: string; runtimeId?: string }>
tiles?: TileTranscriptTarget[]
updateSessionState: (
sessionId: string,
updater: (state: ClientSessionState) => ClientSessionState,
@@ -100,6 +112,13 @@ export async function reconcileTileTranscripts({
) => ClientSessionState
}): Promise<void> {
const tiles = tilesOverride ?? $sessionTiles.get()
const openSignatureKeys = new Set(tiles.map(tileTranscriptSignatureKey))
for (const signatureKey of signatureRef.current.keys()) {
if (!openSignatureKeys.has(signatureKey)) {
signatureRef.current.delete(signatureKey)
}
}
for (const tile of tiles) {
const storedSessionId = tile.storedSessionId
@@ -110,7 +129,7 @@ export async function reconcileTileTranscripts({
continue
}
if (!storedSessionId || !runtimeSessionId || busyRef.current) {
if (!storedSessionId || !runtimeSessionId || tileRuntimeOwnsLiveState(runtimeSessionId)) {
continue
}
@@ -123,23 +142,32 @@ export async function reconcileTileTranscripts({
// With a tiles override (test path), the live $sessionTiles check can't
// see the synthetic tile — treat override tiles as present.
const stillPresent = tilesOverride
? tilesOverride.some(t => t.storedSessionId === storedSessionId && t.runtimeId === runtimeSessionId)
: $sessionTiles.get().some(t => t.storedSessionId === storedSessionId && t.runtimeId === runtimeSessionId)
const tileStillPresent = () =>
tilesOverride
? tilesOverride.some(t => t.storedSessionId === storedSessionId && t.runtimeId === runtimeSessionId)
: $sessionTiles.get().some(t => t.storedSessionId === storedSessionId && t.runtimeId === runtimeSessionId)
// Bot tiles are pinned to an exact owner (connection + target profile);
// read from that backend, not whichever profile is foreground. Tiles
// without a route keep the legacy local read.
const profileScope: ProfileScope = tile.ownerRoute
? { connectionId: tile.ownerRoute.connectionId, profile: tile.ownerRoute.targetProfile ?? tile.ownerRoute.profile }
: undefined
const signatureKey = tileTranscriptSignatureKey(tile)
try {
const latest = await getLatestSessionMessages(storedSessionId)
const latest = await getLatestSessionMessages(storedSessionId, profileScope)
if (requestId !== requestSequenceRef.current || busyRef.current || !stillPresent) {
if (requestId !== requestSequenceRef.current || tileRuntimeOwnsLiveState(runtimeSessionId) || !tileStillPresent()) {
// Tile closed or superseded mid-read — discard AND prune its
// signature so the map doesn't grow one entry per ever-opened tile
// for the app's lifetime (#94255 review point 3).
signatureRef.current.delete(`tile:${storedSessionId}`)
signatureRef.current.delete(signatureKey)
continue
}
const signatureKey = `tile:${storedSessionId}`
const signature = sessionMessagesSignature(latest.messages)
if (signatureRef.current.get(signatureKey) === signature) {
@@ -547,10 +575,8 @@ export function useBackgroundSync({
// transcript signatures, so no-change ticks and closed tiles cost nothing.
const tileRequestSequenceRef = useRef(0)
const tileSignatureRef = useRef(new Map<string, string>())
// Read $busy.get() directly inside the reconcile loop instead of mirroring
// the atom into a ref (lint: no-restricted-syntax — refs synced from atoms
// lag one render). The reconcile runs on tick, not render, so .get() is
// always current.
// Tile reconciliation reads each runtime's live state directly from
// $sessionStates; the primary chat's $busy atom has no authority over tiles.
const requestActiveTranscriptRefresh = useCallback(
(preservePending: boolean) => {
@@ -624,6 +650,19 @@ export function useBackgroundSync({
}
}, [activeConnectionId, activeGatewayProfile, gatewayState, refreshCurrentModel, refreshSessions, requestGateway])
// Reconnect backstop (#94779): turns that finished while the socket was
// down never replay their sessions.changed tick, so the open transcript
// stayed stale until the user reopened it. Pull one signature-gated tail on
// every (re)connect — a no-change read costs nothing. Keyed on the
// connection, not the session, so a plain session switch adds no read;
// messaging transcripts already refresh on open in their own effect below.
useEffect(() => {
if (gatewayState === 'open' && !activeIsMessaging && activeSessionId && activeStoredSessionId) {
requestActiveTranscriptRefresh(true)
}
// eslint-disable-next-line react-hooks/exhaustive-deps -- connect-scoped: session deps would fire on every switch
}, [activeConnectionId, activeGatewayProfile, gatewayState])
// A reconnect loses renderer-only working/attention atoms while the backend
// keeps the actual turns alive. Re-seed from the gateway's in-memory session
// registry immediately, then re-pull on every sessions.changed broadcast; a
@@ -691,17 +730,17 @@ export function useBackgroundSync({
lastRunAt = Date.now()
void refreshSessions()
void refreshMessagingSessions()
// The project tree is a grouping of the same stored rows, so a session
// created/deleted/renamed/re-homed outside this window goes stale in the
// Projects sidebar without this (#100354). refreshProjectTree() keeps the
// cached tree on failure, so a not-yet-ready backend costs nothing.
void refreshProjectTree()
requestActiveTranscriptRefresh(true)
// Bot canonical chats live in workspace tiles, never in the main-pane
// selection — without this they never see background deliveries
// (#93942 scenario A). Signature-gated per tile, so no-change ticks
// cost nothing.
void reconcileTileTranscripts({
busyRef: {
get current() {
return $busy.get()
}
},
requestSequenceRef: tileRequestSequenceRef,
signatureRef: tileSignatureRef,
updateSessionState
@@ -80,6 +80,8 @@ describe('useDesktopIntegrations', () => {
locationPathname = '/',
profileReady = false,
resumeExhaustedSessionId = null as string | null,
// null = config record still loading (the hook takes undefined; null dodges the destructuring default).
resumeLastSession = true as boolean | null,
routedSessionId = null as string | null,
sessions = [] as readonly SessionInfo[]
} = {}) {
@@ -89,6 +91,7 @@ describe('useDesktopIntegrations', () => {
locationPathname,
profileReady,
resumeExhaustedSessionId,
resumeLastSession,
routedSessionId,
sessions
}: {
@@ -96,6 +99,7 @@ describe('useDesktopIntegrations', () => {
locationPathname: string
profileReady: boolean
resumeExhaustedSessionId: string | null
resumeLastSession: boolean | null
routedSessionId: string | null
sessions: readonly SessionInfo[]
}) =>
@@ -108,6 +112,7 @@ describe('useDesktopIntegrations', () => {
profileReady,
refreshSessions: vi.fn(),
resumeExhaustedSessionId,
resumeLastSession: resumeLastSession ?? undefined,
routedSessionId,
runtimeIdByStoredSessionId: { current: new Map() },
sessions
@@ -118,6 +123,7 @@ describe('useDesktopIntegrations', () => {
locationPathname,
profileReady,
resumeExhaustedSessionId,
resumeLastSession,
routedSessionId,
sessions
}
@@ -171,6 +177,7 @@ describe('useDesktopIntegrations', () => {
locationPathname: '/',
profileReady: true,
resumeExhaustedSessionId: null,
resumeLastSession: true,
routedSessionId: null,
sessions: [session({ id: 'remembered-session', profile: 'default' })]
})
@@ -179,6 +186,53 @@ describe('useDesktopIntegrations', () => {
})
})
describe('display.resume_last_session', () => {
it('stays on the fresh chat when the setting is off, and keeps remembering the open chat', () => {
window.localStorage.setItem('hermes.desktop.lastRoute.profile.default', '/remembered-session')
window.localStorage.setItem('hermes.desktop.lastSessionId.profile.default', 'remembered-session')
const sessions = [session({ id: 'remembered-session', profile: 'default' })]
const result = render({ profileReady: true, resumeLastSession: false, sessions })
expect(navigate).not.toHaveBeenCalled()
// The user opens another chat: it is still remembered for the next launch
// (and for notifications), so flipping the switch back on resumes it.
result.rerender({
activeProfile: 'default',
locationPathname: '/other-session',
profileReady: true,
resumeExhaustedSessionId: null,
resumeLastSession: false,
routedSessionId: 'other-session',
sessions: [...sessions, session({ id: 'other-session', profile: 'default' })]
})
expect(window.localStorage.getItem('hermes.desktop.lastSessionId.profile.default')).toBe('other-session')
})
it('holds the restore until the config record answers, then restores when on', () => {
window.localStorage.setItem('hermes.desktop.lastSessionId.profile.default', 'remembered-session')
const sessions = [session({ id: 'remembered-session', profile: 'default' })]
const result = render({ profileReady: true, resumeLastSession: null, sessions })
expect(navigate).not.toHaveBeenCalled()
result.rerender({
activeProfile: 'default',
locationPathname: '/',
profileReady: true,
resumeExhaustedSessionId: null,
resumeLastSession: true,
routedSessionId: null,
sessions
})
expect(navigate).toHaveBeenCalledWith('/remembered-session', { replace: true })
})
})
describe('ownership validation', () => {
it('refuses to restore a session route owned by another profile', () => {
window.localStorage.setItem('hermes.desktop.lastRoute.profile.default', '/ai-session')
@@ -329,6 +383,7 @@ describe('useDesktopIntegrations', () => {
locationPathname: '/ops-session',
profileReady: true,
resumeExhaustedSessionId: null,
resumeLastSession: true,
routedSessionId: 'ops-session',
sessions
})
@@ -395,6 +450,7 @@ describe('useDesktopIntegrations', () => {
locationPathname: '/settings',
profileReady: true,
resumeExhaustedSessionId: null,
resumeLastSession: true,
routedSessionId: null,
sessions: []
})
@@ -41,6 +41,8 @@ interface DesktopIntegrationsParams {
navigate: (to: string, options?: { replace?: boolean }) => void
profileReady: boolean
refreshSessions: () => Promise<unknown> | unknown
/** `display.resume_last_session`; `undefined` while the config record is still loading. */
resumeLastSession: boolean | undefined
resumeExhaustedSessionId: null | string
routedSessionId: null | string
runtimeIdByStoredSessionId: { readonly current: Map<string, string> }
@@ -60,6 +62,7 @@ export function useDesktopIntegrations({
navigate,
profileReady,
refreshSessions,
resumeLastSession,
resumeExhaustedSessionId,
routedSessionId,
runtimeIdByStoredSessionId,
@@ -73,7 +76,12 @@ export function useDesktopIntegrations({
// Background MCP health: HTTP/SSE servers only (never spawns stdio),
// notifies on transitions into needs-auth/error with a Sign in action.
startMcpHealthChecker()
const unsubscribe = window.hermesDesktop?.onOpenUpdatesRequested?.(() => openUpdatesWindow())
// The native "Check for Updates…" menu item lives in the app menu next to
// "About Hermes" — it is the OS-standard affordance for updating THIS app,
// so it always opens the client overlay. Inheriting the connection-mode
// default pointed a Mac at its remote Linux backend and left the app itself
// silently stale (#70266).
const unsubscribe = window.hermesDesktop?.onOpenUpdatesRequested?.(() => openUpdatesWindow('client'))
return () => {
unsubscribe?.()
@@ -105,6 +113,20 @@ export function useDesktopIntegrations({
// Only cold-start navigation at the default route is replaceable; a deep
// link or hidden-then-shown window keeps its explicit destination.
if (locationPathname === NEW_CHAT_ROUTE) {
// display.resume_last_session (#60812): hold the latch until the config
// record answers, then either restore below or stay on the fresh chat.
// Remembered ids keep being written either way, so flipping the switch
// back on resumes from the very next launch.
if (resumeLastSession === undefined) {
return
}
if (!resumeLastSession) {
restoredRef.current = true
return
}
const route = getRememberedRoute(activeProfile)
const routeSession = route ? routeSessionId(route) : null
const last = getRememberedSessionId(activeProfile)
@@ -163,7 +185,7 @@ export function useDesktopIntegrations({
} else if (!routedSessionId && !isOverlayView(appViewForPath(locationPathname))) {
setRememberedRoute(locationPathname, activeProfile)
}
}, [activeProfile, locationPathname, navigate, profileReady, routedSessionId, sessions])
}, [activeProfile, locationPathname, navigate, profileReady, resumeLastSession, routedSessionId, sessions])
useEffect(() => {
if (!profileReady || !resumeExhaustedSessionId) {
@@ -38,6 +38,7 @@ vi.mock('@/store/session', async importActual => ({
const { createSessionRpcDispatcher } = await import('./session-rpc-dispatcher')
const { $connectionsRegistry } = await import('@/store/connection-registry-state')
const { $profiles } = await import('@/store/profile')
const { $removedSessionIds, $sessionMutationsInFlight } = await import('@/store/projects')
const { _resetSessionOwnerHintsForTests, setCronSessions, setMessagingSessions, setSessionOwnerHint, setSessions } =
await import('@/store/session')
@@ -75,6 +76,8 @@ afterEach(() => {
setMessagingSessions([])
$sessionTiles.set([])
$profiles.set([])
$removedSessionIds.set(new Set())
$sessionMutationsInFlight.set(new Set())
_resetSessionOwnerHintsForTests({ storage: true })
sessionMocks.requestSessionResume.mockReset()
vi.clearAllMocks()
@@ -233,6 +236,22 @@ describe('createSessionRpcDispatcher: stale runtime recovery', () => {
expect(sessionMocks.requestSessionResume).not.toHaveBeenCalled()
})
it.each([
['tombstoned', $removedSessionIds],
['being deleted', $sessionMutationsInFlight]
])('does not rebind a selected session that is %s', async (_state, sessions) => {
setSessions([makeSessionInfo({ connection_id: 'local', id: 'stored-omar', profile: 'omar' })])
sessions.set(new Set(['stored-omar']))
gatewayMocks.requestGatewayForAgent.mockRejectedValueOnce(
Object.assign(new Error('session not found'), { code: 4001 })
)
const { request } = dispatcher(undefined, 'stored-omar')
await expect(request('process.list', { session_id: 'rt-omar' })).rejects.toThrow('session not found')
expect(sessionMocks.requestSessionResume).not.toHaveBeenCalled()
})
it('does not interpret an unrelated coded RPC failure as a stale runtime', async () => {
setSessions([makeSessionInfo({ connection_id: 'local', id: 'stored-omar', profile: 'omar' })])
gatewayMocks.requestGatewayForAgent.mockRejectedValueOnce(
@@ -35,6 +35,7 @@ import type { MutableRefObject } from 'react'
import { resolveSessionOwner } from '@/app/session/hooks/use-session-actions/utils'
import type { ClientSessionState } from '@/app/types'
import { $removedSessionIds, $sessionMutationsInFlight } from '@/store/projects'
import { isSessionGoneForBackgroundPolling } from '@/store/runtime-gone'
import { getSessionOwnerHint, knownSessionOwner, ownerLookupSessionRows, requestSessionResume } from '@/store/session'
import { assertSessionOwnerResolved } from '@/store/session-owner-resolution'
@@ -112,6 +113,8 @@ export function createSessionRpcDispatcher(deps: SessionRpcDispatcherDeps): Ambi
paramSessionId &&
routingSessionId &&
routingSessionId === selectedStoredSessionIdRef.current &&
!$removedSessionIds.get().has(routingSessionId) &&
!$sessionMutationsInFlight.get().has(routingSessionId) &&
isSessionGoneForBackgroundPolling(error)
) {
requestSessionResume(routingSessionId, typeof owner === 'object' && owner ? owner : undefined)
+11
View File
@@ -93,6 +93,7 @@ import { CommandPalette } from '../command-palette'
import { triggerAndRefreshCronJobs } from '../cron/cron-actions'
import { useGatewayBoot } from '../gateway/hooks/use-gateway-boot'
import { useGatewayRequest } from '../gateway/hooks/use-gateway-request'
import { useHermesConfigRecord } from '../hooks/use-config-record'
import { useKeybinds } from '../hooks/use-keybinds'
import { useHudHandoff } from '../hud/handoff'
import { ModelPickerOverlay } from '../model-picker-overlay'
@@ -842,6 +843,15 @@ export function ContribWiring({ children }: { children: ReactNode }) {
// remembered-session restore, and cross-window session-list sync.
const previewTarget = useStore($previewTarget)
// display.resume_last_session gates the cold-start restore. `undefined` while
// the record is still loading holds the restore latch open; a failed fetch
// falls back to the historical behavior (resume).
const configRecord = useHermesConfigRecord()
const resumeLastSession = configRecord.isPending
? undefined
: (configRecord.data?.display as { resume_last_session?: unknown } | undefined)?.resume_last_session !== false
useDesktopIntegrations({
activeProfile: normalizeProfileKey(activeGatewayProfile),
chatOpen,
@@ -850,6 +860,7 @@ export function ContribWiring({ children }: { children: ReactNode }) {
navigate,
profileReady: boot.phase === 'renderer.ready',
refreshSessions,
resumeLastSession,
resumeExhaustedSessionId,
routedSessionId,
runtimeIdByStoredSessionId: runtimeIdByStoredSessionIdRef,
@@ -44,6 +44,7 @@ import {
isCurrentGatewaySwitch,
registerGatewaySwitchLifecycle
} from '@/store/gateway-switch'
import { checkLocalRuntimeUpdate, watchLocalRuntimeJobs } from '@/store/local-runtime-jobs'
import { notify, notifyError } from '@/store/notifications'
import {
$activeGatewayProfile,
@@ -663,6 +664,12 @@ export function useGatewayBoot({
completeDesktopBoot()
bootCompleted = true
// Rediscover local-runtime jobs (model downloads, runtime installs)
// that were running before a reload — the backend registry is the
// authority; this just resumes following it.
watchLocalRuntimeJobs()
// One-per-session engine-update pointer (enabled runtimes only).
void checkLocalRuntimeUpdate()
} catch (err) {
const mayPublishFailure =
!cancelled && (switchToken === null ? !$gatewaySwitching.get() : isCurrentGatewaySwitch(switchToken))
+4 -91
View File
@@ -13,9 +13,9 @@ import { useHudClickThrough } from './click-through'
import { useHudGameOverlay } from './game-overlay'
import { useHudGlass } from './glass'
import { useHudGoto, useReportHudSession } from './handoff'
import { hudTranscriptHeight } from './layout'
import { hudResizeDirections, useHudResizeHandle } from './resize-handle'
import { useHudThreadFocus } from './thread-focus'
import { useHudTranscriptBand } from './transcript-band'
/** How long the transcript lingers at its glanceable opacity — after a turn
* lands, or after you let go of the composer — before it goes. This is the ONLY
@@ -39,11 +39,6 @@ const HUD_DIM_MS = Math.round(HUD_FADE_MS * 1.5)
* drawn down into the bar rather than the two dissolving in lockstep. */
const HUD_COLLAPSE_MS = Math.round(HUD_FADE_MS * 0.66)
/** Breathing room the sheet keeps above the first row, so the fade has
* somewhere to land. Folded into the measured height rather than added in CSS,
* so an empty transcript measures a true zero instead of a 12px strip. */
const HUD_SHEET_OVERHANG_PX = 12
/** Composer on top, transcript always hanging below it — Spotlight's shape,
* rather than flipping to follow the screen edge the HUD is parked against. */
const HUD_THREAD_ALWAYS_BELOW = true
@@ -275,6 +270,8 @@ export function HudShell() {
}
}, [])
const rootRef = useRef<HTMLDivElement | null>(null)
// Whether bar + band actually cover the window. Gates the frost, which is
// native vibrancy and therefore the WINDOW's content view — it fills the whole
// rectangle and nothing in the page can clip it to the sheet. Whenever the
@@ -282,91 +279,7 @@ export function HudShell() {
// a grey slab hanging under the bar with nothing in it. Now that the band is
// capped it almost never covers the window, so this is almost always false —
// which is correct, and asking anything looser paints the slab back.
const [filled, setFilled] = useState(false)
const rootRef = useRef<HTMLDivElement | null>(null)
useEffect(() => {
const root = rootRef.current
if (!root) {
return
}
let viewport: HTMLElement | null = null
const ro = new ResizeObserver(() => measure())
const measure = () => {
const el = viewport ?? root.querySelector<HTMLElement>('[data-slot="aui_thread-viewport"]')
if (el !== viewport) {
viewport = el
if (el) {
ro.observe(el)
if (el.firstElementChild) {
ro.observe(el.firstElementChild)
}
}
}
// How tall the band actually needs to be — the tight bbox of the message
// rows only. Measuring to the viewport edge counted the full-window scroll
// container (min-height: 100%) as transcript and painted a empty slab almost
// the size of the HUD.
const rows = el?.querySelectorAll<HTMLElement>('[data-slot="aui_thread-content"] > *:not([data-slot])')
// Zero-height rows are not a transcript. A fresh thread still renders
// scaffolding inside the content box (clearance, empty state), so
// counting rows alone paid the overhang for nothing and left a sliver of
// sheet hanging under the bar with no text in it.
const text = !rows?.length
? 0
: Math.max(0, rows[rows.length - 1].getBoundingClientRect().bottom - rows[0].getBoundingClientRect().top)
const contentSpan = text < 1 ? 0 : text + HUD_SHEET_OVERHANG_PX
// Once the HUD has a transcript, a resize must buy readable scrollback.
// The old glance-band ceiling froze this at 152px and turned every extra
// pixel of native window height into empty transparent chrome.
const visible = hudTranscriptHeight({
barHeight: root.querySelector<HTMLElement>('[data-slot="composer-dock"]')?.getBoundingClientRect().height ?? 0,
contentHeight: contentSpan,
viewportHeight: window.innerHeight
})
root.style.setProperty('--hud-band-height', `${visible}px`)
// …and the bar's real height, which is what the thread has to clear.
// --composer-measured-height would be the obvious source, but it is a
// surface var that never lands here, so the clearance silently fell back
// to the root estimate and reserved ~20px more than the bar occupies —
// a visible hole under the last message.
const bar = root.querySelector<HTMLElement>('[data-slot="composer-dock"]')
const barHeight = bar?.getBoundingClientRect().height ?? 0
if (bar) {
ro.observe(bar)
root.style.setProperty('--hud-bar-height', `${Math.round(barHeight)}px`)
}
setFilled(barHeight + visible >= window.innerHeight - 1)
}
// The viewport mounts async (lazy chat surface); poll briefly until it
// exists, then let the ResizeObserver own it. Window resize is separate:
// the transcript's rows may not change size, but the available scrollback
// must, so observing the rows alone cannot update the band.
measure()
const probe = setInterval(measure, 500)
window.addEventListener('resize', measure)
return () => {
clearInterval(probe)
window.removeEventListener('resize', measure)
ro.disconnect()
}
}, [])
const filled = useHudTranscriptBand(rootRef)
useHudGlass(rootRef, filled)
useHudClickThrough(rootRef)
@@ -0,0 +1,64 @@
import { act, render } from '@testing-library/react'
import { useRef } from 'react'
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
import { stubResizeObserver } from '@/test/jsdom'
import { useHudTranscriptBand } from './transcript-band'
function Harness({ withViewport }: { withViewport: boolean }) {
const ref = useRef<HTMLDivElement | null>(null)
useHudTranscriptBand(ref)
return (
<div ref={ref}>
<div data-slot="composer-dock" />
{withViewport && (
<div data-slot="aui_thread-viewport">
<div data-slot="aui_thread-content">
<div>row</div>
</div>
</div>
)}
</div>
)
}
beforeEach(() => {
stubResizeObserver()
vi.useFakeTimers()
})
afterEach(() => {
vi.useRealTimers()
})
describe('useHudTranscriptBand', () => {
// The bug this replaced: the probe polled every 500ms for the lifetime of
// the HUD window, duplicating every measurement the ResizeObserver already
// owned once the viewport existed — a permanent idle timer firing re-renders
// forever instead of the "poll briefly, then hand off" the code documented.
it('stops polling once the viewport mounts', () => {
const measureSpy = vi.spyOn(HTMLElement.prototype, 'getBoundingClientRect')
const { rerender } = render(<Harness withViewport={false} />)
const beforeWaiting = measureSpy.mock.calls.length
act(() => vi.advanceTimersByTime(500))
act(() => vi.advanceTimersByTime(500))
const whileWaiting = measureSpy.mock.calls.length
expect(whileWaiting).toBeGreaterThan(beforeWaiting)
rerender(<Harness withViewport />)
act(() => vi.advanceTimersByTime(500))
const justAfterFound = measureSpy.mock.calls.length
expect(justAfterFound).toBeGreaterThan(whileWaiting)
act(() => vi.advanceTimersByTime(10_000))
const muchLater = measureSpy.mock.calls.length
expect(muchLater).toBe(justAfterFound)
})
})
+117
View File
@@ -0,0 +1,117 @@
import { type RefObject, useEffect, useState } from 'react'
import { hudTranscriptHeight } from './layout'
/** Breathing room the sheet keeps above the first row, so the fade has
* somewhere to land. Folded into the measured height rather than added in CSS,
* so an empty transcript measures a true zero instead of a 12px strip. */
const HUD_SHEET_OVERHANG_PX = 12
/**
* Measures the HUD's transcript band and publishes it as `--hud-band-height` /
* `--hud-bar-height` on the root, returning whether the band + bar fill the
* window (which gates the frost — see `useHudGlass`).
*
* The viewport mounts async (lazy chat surface); poll briefly until it exists,
* then let the ResizeObserver own it. Window resize is separate: the
* transcript's rows may not change size, but the available scrollback must, so
* observing the rows alone cannot update the band.
*/
export function useHudTranscriptBand(rootRef: RefObject<HTMLDivElement | null>): boolean {
const [filled, setFilled] = useState(false)
useEffect(() => {
const root = rootRef.current
if (!root) {
return
}
let viewport: HTMLElement | null = null
const ro = new ResizeObserver(() => measure())
const measure = () => {
const el = viewport ?? root.querySelector<HTMLElement>('[data-slot="aui_thread-viewport"]')
if (el !== viewport) {
viewport = el
if (el) {
ro.observe(el)
if (el.firstElementChild) {
ro.observe(el.firstElementChild)
}
}
}
// How tall the band actually needs to be — the tight bbox of the message
// rows only. Measuring to the viewport edge counted the full-window scroll
// container (min-height: 100%) as transcript and painted a empty slab almost
// the size of the HUD.
const rows = el?.querySelectorAll<HTMLElement>('[data-slot="aui_thread-content"] > *:not([data-slot])')
// Zero-height rows are not a transcript. A fresh thread still renders
// scaffolding inside the content box (clearance, empty state), so
// counting rows alone paid the overhang for nothing and left a sliver of
// sheet hanging under the bar with no text in it.
const text = !rows?.length
? 0
: Math.max(0, rows[rows.length - 1].getBoundingClientRect().bottom - rows[0].getBoundingClientRect().top)
const contentSpan = text < 1 ? 0 : text + HUD_SHEET_OVERHANG_PX
// Once the HUD has a transcript, a resize must buy readable scrollback.
// The old glance-band ceiling froze this at 152px and turned every extra
// pixel of native window height into empty transparent chrome.
const visible = hudTranscriptHeight({
barHeight: root.querySelector<HTMLElement>('[data-slot="composer-dock"]')?.getBoundingClientRect().height ?? 0,
contentHeight: contentSpan,
viewportHeight: window.innerHeight
})
root.style.setProperty('--hud-band-height', `${visible}px`)
// …and the bar's real height, which is what the thread has to clear.
// --composer-measured-height would be the obvious source, but it is a
// surface var that never lands here, so the clearance silently fell back
// to the root estimate and reserved ~20px more than the bar occupies —
// a visible hole under the last message.
const bar = root.querySelector<HTMLElement>('[data-slot="composer-dock"]')
const barHeight = bar?.getBoundingClientRect().height ?? 0
if (bar) {
ro.observe(bar)
root.style.setProperty('--hud-bar-height', `${Math.round(barHeight)}px`)
}
setFilled(barHeight + visible >= window.innerHeight - 1)
}
measure()
// Once the viewport has mounted, the ResizeObserver above owns every
// future measurement — a probe that never stops re-runs this on every
// tick forever, which is exactly the sustained idle CPU / re-render loop
// the HUD must not have.
const probe = window.setInterval(() => {
if (viewport) {
window.clearInterval(probe)
return
}
measure()
}, 500)
window.addEventListener('resize', measure)
return () => {
window.clearInterval(probe)
window.removeEventListener('resize', measure)
ro.disconnect()
}
}, [rootRef])
return filled
}
@@ -0,0 +1,127 @@
/**
* End-to-end owner routing for BRANCH (the #97764-adjacent strand).
*
* The unit tests in use-session-actions.test.tsx mock `@/store/gateway`, so
* they prove the branch path ASKS for the right route. They cannot prove the
* routing layer HONOURS it. This file mocks nothing inside the router: the real
* `requestGatewayForAgent` runs against a fake Electron bridge + transport, so
* a regression that re-collapses a registry route onto the ambient socket fails
* here even if the call-site assertions still pass.
*
* Reproduces the reported shape: a session owned by a remote connection
* ("pandora") is branched while a different backend is active. Before the fix
* the create rode the ambient socket and the child was created on the wrong
* backend (or nowhere), stranding an optimistic sidebar row on an id no backend
* owned — "Couldn't load this session".
*/
import { beforeEach, describe, expect, it, vi } from 'vitest'
// Every socket the registry dials, and every RPC that travelled over one.
const dialed: { connectionId: string; profile: string }[] = []
const sent: { method: string; params: Record<string, unknown>; url: string }[] = []
class FakeHermesGateway {
connectionState = 'closed'
private url = ''
async connect(wsUrl: string) {
if (typeof wsUrl !== 'string' || !wsUrl.startsWith('ws')) {
throw new Error(`bad ws url: ${String(wsUrl)}`)
}
this.url = wsUrl
this.connectionState = 'open'
}
async request<T>(method: string, params: Record<string, unknown> = {}): Promise<T> {
sent.push({ method, params, url: this.url })
if (method === 'session.create' || method === 'session.branch') {
return { session_id: 'branch-runtime', stored_session_id: 'branch-stored' } as T
}
return {} as T
}
close() {
this.connectionState = 'closed'
}
onEvent(_listener: (event: unknown) => void) {
return () => undefined
}
onState(_listener: (state: unknown) => void) {
return () => undefined
}
onStateChange(_listener: (state: unknown) => void) {
return () => undefined
}
on() {}
off() {}
addEventListener() {}
removeEventListener() {}
}
vi.mock('@/hermes', async importOriginal => ({
...(await importOriginal<Record<string, unknown>>()),
HermesGateway: FakeHermesGateway,
setApiRequestConnection: vi.fn()
}))
describe('branch owner routing (real router, faked transport)', () => {
beforeEach(() => {
dialed.length = 0
sent.length = 0
vi.resetModules()
// A registry with two backends exposing the SAME profile name — the exact
// ambiguity that makes profile-only routing wrong.
;(window as unknown as { hermesDesktop: unknown }).hermesDesktop = {
getConnection: async () => ({ mode: 'local' }),
getConnectionFor: async ({ connectionId, profile }: { connectionId: string; profile: string }) => {
dialed.push({ connectionId, profile })
return { connectionId, mode: 'remote', profile }
},
getGatewayWsUrlFor: async ({ connectionId, profile }: { connectionId: string; profile: string }) =>
`ws://${connectionId}/gateway?profile=${profile}`,
touchBackend: async () => undefined
}
})
it('dials the parent connection and sends the create over that socket', async () => {
const { requestGatewayForAgent } = await import('@/store/gateway')
await requestGatewayForAgent('pandora', 'default', 'session.create', {
parent_session_id: 'stored-parent',
source: 'desktop'
})
// The registry resolved a socket for the PARENT's connection...
expect(dialed).toContainEqual({ connectionId: 'pandora', profile: 'default' })
// ...and the create actually travelled over that socket.
const create = sent.find(entry => entry.method === 'session.create')
expect(create).toBeDefined()
expect(create!.url).toContain('pandora')
expect(create!.params).toMatchObject({ parent_session_id: 'stored-parent' })
})
it('keeps two same-named profiles on separate sockets', async () => {
const { requestGatewayForAgent } = await import('@/store/gateway')
await requestGatewayForAgent('pandora', 'default', 'session.create', { source: 'desktop' })
await requestGatewayForAgent('other-box', 'default', 'session.create', { source: 'desktop' })
const urls = sent.filter(entry => entry.method === 'session.create').map(entry => entry.url)
expect(urls).toHaveLength(2)
// Same profile name, different backends — they must NOT share a socket.
expect(new Set(urls).size).toBe(2)
expect(urls.some(url => url.includes('pandora'))).toBe(true)
expect(urls.some(url => url.includes('other-box'))).toBe(true)
})
})

Some files were not shown because too many files have changed in this diff Show More