Merge origin/main into core-tool-deferral (resolve show_tip test seam onto the check_tips_enabled gate)

This commit is contained in:
Teknium
2026-09-01 21:49:14 -07:00
1137 changed files with 138764 additions and 7075 deletions
+3 -3
View File
@@ -107,9 +107,9 @@ plans/
.hadolint.yaml
.mailmap
# Repo-root debug/export artifacts — must never reach image layers (COPY . .)
# Debug/export artifacts — must never reach image layers (COPY . .)
/log.txt
/sqlite_leak_fix.png
/*.png.bak
/default.tar.gz
/*.tar.gz
*.tar.gz
*.tgz
+7 -5
View File
@@ -322,12 +322,14 @@ BROWSERBASE_PROXIES=true
# Uses custom Chromium build to avoid bot detection altogether
BROWSERBASE_ADVANCED_STEALTH=false
# Browser engine for local mode (default: auto = Chrome)
# "auto" — use Chrome (don't pass --engine flag)
# "lightpanda" — use Lightpanda (1.3-5.8x faster navigation, no screenshots)
# Local browser engine (default: auto = Chrome)
# "auto" — use Chrome
# "lightpanda" — use Lightpanda (faster navigation, no screenshots)
# "chrome" — explicitly request Chrome
# Requires agent-browser v0.25.3+. Lightpanda commands that fail or return
# empty results are automatically retried with Chrome.
# Browser Use mode (default) spawns `lightpanda serve` itself; the built-in
# browser tools pass --engine to agent-browser v0.25.3+ and retry failed or
# empty Lightpanda results with Chrome. Ignored while a cloud provider,
# Camofox or browser.cdp_url is active (`hermes doctor` reports that).
# Also configurable via browser.engine in config.yaml.
# AGENT_BROWSER_ENGINE=auto
@@ -0,0 +1,33 @@
name: Case Collision Check
# Rejects PRs that track two files whose paths differ only by case
# (README.md vs readme.md, src/Foo.py vs SRC/foo.py).
#
# Linux is case-sensitive; Windows and macOS (default) are not. A
# case-colliding pair lives fine in a Linux checkout and silently breaks
# every clone on a case-insensitive host — the filesystem can hold only
# one of them, so checkout fails or whichever wins overwrites the other.
# Git won't prevent the pair from landing (it only warns at checkout time,
# on a case-insensitive FS, for the client doing the checkout), so the only
# enforcement point is CI, on Linux, against the index.
#
# Runs unconditionally (no change-classifier gate): a collision can ship in
# any kind of PR — docs, JS, config, not just Python — so gating on a
# language lane would be the same "passive rule that cannot enforce a
# policy" trap the infographic check exists to close.
on:
workflow_call:
permissions:
contents: read
jobs:
check-case-collisions:
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Run case-collision checker
run: python3 scripts/check-case-collisions.py
+20 -8
View File
@@ -122,14 +122,14 @@ jobs:
# Tests-only PRs (~17% of commits) skip this 5-minute job — the longest
# single job in the workflow — while still running the full pytest lanes.
#
# ⛔ TEMPORARILY DISABLED (Aug 2, 2026, Teknium) — the suite is red on
# every PR and on main itself since the Aug 1 night engines/npm churn
# (#76499 → #76562 → #76575): the mock-backend Electron window never
# gets a title, so boot/chat/setup/interim specs all fail identically
# regardless of the PR's diff (verified on #76573 and the docs-only
# #76582). Tracking issue: #76627 (assigned: Ari). To re-enable,
# delete the `false &&` below — nothing else changed.
if: ${{ false && (needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true') }}
# Re-disabled (Sep 2026): the Sep 1 re-enable is still incredibly flaky.
# Keep this a bare `if: false`. The earlier
# `${{ false && (... || ...) }}` form on this reusable-workflow job made
# GitHub's workflow parser fail at startup ("An unexpected error has
# occurred") — every ci.yaml run repo-wide dispatched 0 jobs from
# 24f5a60ed1 until this line changed. To re-enable, restore:
# if: ${{ needs.detect.outputs.python_prod == 'true' || needs.detect.outputs.frontend == 'true' }}
if: false
uses: ./.github/workflows/e2e-desktop.yml
docs-site:
@@ -166,6 +166,16 @@ jobs:
needs: detect
uses: ./.github/workflows/infographic-check.yml
profile-artifact-check:
name: Profile artifact check
needs: detect
uses: ./.github/workflows/profile-artifact-check.yml
case-collision-check:
name: Check no case-colliding filenames
needs: detect
uses: ./.github/workflows/case-collision-check.yml
lockfile-diff:
name: package-lock.json diff
needs: detect
@@ -228,8 +238,10 @@ jobs:
- history-check
- contributor-check
- uv-lockfile
- case-collision-check
- lockfile-diff
- docker-lint
- profile-artifact-check
- supply-chain
- review-labels
- osv-scanner
+6
View File
@@ -94,6 +94,9 @@ jobs:
- name: Checkout code
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Reject profile exports in the build context
run: python3 scripts/ci/check_profile_archive_boundary.py
# Retry once on transient Docker Hub / buildkit pull failures
# (connection reset, auth token timeout, rate limiting). The action
# generates a unique builder name per invocation so the retry doesn't
@@ -206,6 +209,9 @@ jobs:
- name: Checkout trusted source
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Reject profile exports in the build context
run: python3 scripts/ci/check_profile_archive_boundary.py
# Retry once on transient Docker Hub / buildkit pull failures.
# See build job for rationale; same pattern.
- name: Set up Docker Buildx
@@ -0,0 +1,21 @@
name: Profile Artifact Boundary
# A reusable, unconditional guard for the incident class in #92457. Ignore
# files reduce accidental staging; this job is the enforcement boundary that
# still catches `git add -f` and generated files present during a build.
on:
workflow_call:
permissions:
contents: read
jobs:
check-profile-artifacts:
name: Reject profile archives
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Reject profile exports in the checkout
run: python3 scripts/ci/check_profile_archive_boundary.py
+20 -1
View File
@@ -50,7 +50,7 @@ jobs:
- name: Install dependencies
uses: ./.github/actions/retry
with:
command: uv sync --locked --python 3.11 --extra dev
command: uv sync --locked --python 3.11 --extra dev --extra messaging
- name: Run venv-holder live E2E
shell: bash
@@ -58,4 +58,23 @@ jobs:
set -uo pipefail
uv run --no-sync python -m pytest \
tests/hermes_cli/test_venv_holder_windows_live.py \
tests/hermes_cli/test_taskkill_identity_windows_live.py \
tests/hermes_cli/test_git_trampoline_windows_live.py \
"tests/hermes_cli/test_managed_uv.py::TestWindowsRuntimeSelfLock" \
-o addopts= -v -p no:cacheprovider
- name: Run Telegram CLOSE-WAIT reconnect live E2E (#87057)
shell: bash
run: |
set -uo pipefail
uv run --no-sync python -m pytest \
tests/gateway/test_telegram_closewait_windows_live.py \
-o addopts= -v -p no:cacheprovider
- name: Run background-executor spawn parity live E2E (#70716)
shell: bash
run: |
set -uo pipefail
uv run --no-sync python -m pytest \
tests/tools/test_process_registry_windows_live.py \
-o addopts= -v -p no:cacheprovider
+6 -2
View File
@@ -99,11 +99,12 @@ apps/desktop/src/**/*.d.ts
!apps/desktop/src/global.d.ts
!apps/desktop/src/vite-env.d.ts
# Repo-root build/debug artifacts that must never be committed
# Build/debug artifacts that must never be committed
/log.txt
/sqlite_leak_fix.png
/*.png.bak
/default.tar.gz
*.tar.gz
*.tgz
apps/shared/src/**/*.js
apps/shared/src/**/*.js.map
apps/shared/src/**/*.d.ts
@@ -213,3 +214,6 @@ native/fts5_cjk/*.so
# by accident via 3a69e34702, removed in the #72002 salvage).
.lazy-refresh-incomplete
.skills_prompt_snapshot.json
# Disposable profile created by scripts/probe_active_session_exclusivity.py
.probe-home/
+131 -60
View File
@@ -1347,8 +1347,8 @@ def init_agent(
elif base_url_host_matches(effective_base, "portal.qwen.ai"):
client_kwargs["default_headers"] = _ra()._qwen_portal_headers()
elif base_url_host_matches(effective_base, "chatgpt.com"):
from agent.auxiliary_client import _codex_cloudflare_headers
client_kwargs["default_headers"] = _codex_cloudflare_headers(
from agent.codex_headers import codex_cloudflare_headers
client_kwargs["default_headers"] = codex_cloudflare_headers(
api_key, base_url=effective_base,
)
elif base_url_host_matches(effective_base, "x.ai"):
@@ -1390,14 +1390,71 @@ def init_agent(
if _routed_headers:
client_kwargs["default_headers"] = dict(_routed_headers)
else:
# When the user explicitly chose a non-OpenRouter provider
# but no credentials were found, fail fast with a clear
# message instead of silently routing through OpenRouter.
# No credentials resolved for the configured provider. Give
# the user-configured fallback chain a chance BEFORE failing
# (#17929) — regardless of WHICH provider failed. An
# exhausted single-entry pool (typically ``openrouter``
# under free-tier daily quotas) must still reach the chain
# instead of dying at init with a misleading "No LLM
# provider configured" error. Only providers explicitly
# chosen by name keep the dedicated missing-key diagnostic.
_explicit = (agent.provider or "").strip().lower()
if _explicit and _explicit not in {"auto", "openrouter", "custom"}:
# Look up the actual env var name from the provider
# config — some providers use non-standard names
# (e.g. alibaba → DASHSCOPE_API_KEY, not ALIBABA_API_KEY).
# --- Init-time fallback (#17929) ---
_fb_entries = []
if isinstance(fallback_model, list):
_fb_entries = [
f for f in fallback_model
if isinstance(f, dict) and f.get("provider") and f.get("model")
]
elif isinstance(fallback_model, dict) and fallback_model.get("provider") and fallback_model.get("model"):
_fb_entries = [fallback_model]
_fb_resolved = False
for _fb in _fb_entries:
try:
from hermes_cli.fallback_config import resolve_entry_api_key
_fb_explicit_key = resolve_entry_api_key(_fb)
_fb_client, _fb_model = resolve_provider_client(
_fb["provider"], model=_fb["model"], raw_codex=True,
explicit_base_url=_fb.get("base_url"),
explicit_api_key=_fb_explicit_key,
)
except Exception as _fb_exc:
logger.debug(
"Init-time fallback entry %s failed: %s",
_fb.get("provider"), _fb_exc,
)
continue
if _fb_client is not None:
agent.provider = _fb["provider"]
agent.model = _fb_model or _fb["model"]
agent._fallback_activated = True
client_kwargs = {
"api_key": _fb_client.api_key,
"base_url": str(_fb_client.base_url),
}
if _provider_timeout is not None:
client_kwargs["timeout"] = _provider_timeout
_fb_headers = getattr(_fb_client, "_custom_headers", None)
if not _fb_headers:
_fb_headers = getattr(_fb_client, "default_headers", None)
if not _fb_headers:
_fb_headers = getattr(_fb_client, "_default_headers", None)
if _fb_headers:
client_kwargs["default_headers"] = dict(_fb_headers)
_fb_resolved = True
break
if (
not _fb_resolved
and _explicit
and _explicit not in {"auto", "openrouter", "custom"}
):
# Explicitly chosen non-OpenRouter provider with neither
# credentials nor a usable fallback: fail fast with a
# clear message instead of silently routing through
# OpenRouter. Look up the actual env var name from the
# provider config — some providers use non-standard
# names (e.g. alibaba → DASHSCOPE_API_KEY, not
# ALIBABA_API_KEY).
_env_hint = f"{_explicit.upper()}_API_KEY"
try:
from hermes_cli.auth import PROVIDER_REGISTRY
@@ -1406,56 +1463,11 @@ def init_agent(
_env_hint = _pcfg.api_key_env_vars[0]
except Exception:
pass
# --- Init-time fallback (#17929) ---
_fb_entries = []
if isinstance(fallback_model, list):
_fb_entries = [
f for f in fallback_model
if isinstance(f, dict) and f.get("provider") and f.get("model")
]
elif isinstance(fallback_model, dict) and fallback_model.get("provider") and fallback_model.get("model"):
_fb_entries = [fallback_model]
_fb_resolved = False
for _fb in _fb_entries:
try:
from hermes_cli.fallback_config import resolve_entry_api_key
_fb_explicit_key = resolve_entry_api_key(_fb)
_fb_client, _fb_model = resolve_provider_client(
_fb["provider"], model=_fb["model"], raw_codex=True,
explicit_base_url=_fb.get("base_url"),
explicit_api_key=_fb_explicit_key,
)
except Exception as _fb_exc:
logger.debug(
"Init-time fallback entry %s failed: %s",
_fb.get("provider"), _fb_exc,
)
continue
if _fb_client is not None:
agent.provider = _fb["provider"]
agent.model = _fb_model or _fb["model"]
agent._fallback_activated = True
client_kwargs = {
"api_key": _fb_client.api_key,
"base_url": str(_fb_client.base_url),
}
if _provider_timeout is not None:
client_kwargs["timeout"] = _provider_timeout
_fb_headers = getattr(_fb_client, "_custom_headers", None)
if not _fb_headers:
_fb_headers = getattr(_fb_client, "default_headers", None)
if not _fb_headers:
_fb_headers = getattr(_fb_client, "_default_headers", None)
if _fb_headers:
client_kwargs["default_headers"] = dict(_fb_headers)
_fb_resolved = True
break
if not _fb_resolved:
raise RuntimeError(
f"Provider '{_explicit}' is set in config.yaml but no API key "
f"was found. Set the {_env_hint} environment "
f"variable, or switch to a different provider with `hermes model`."
)
raise RuntimeError(
f"Provider '{_explicit}' is set in config.yaml but no API key "
f"was found. Set the {_env_hint} environment "
f"variable, or switch to a different provider with `hermes model`."
)
if not getattr(agent, "_fallback_activated", False):
# No provider configured — reject with a clear message.
raise RuntimeError(
@@ -2777,6 +2789,34 @@ def init_agent(
if not agent.quiet_mode:
_ra().logger.info("Using context engine: %s", _selected_engine.name)
else:
# Native Gemini output reservation (#57275 claim 4): when
# model.max_tokens is unset, the native generateContent adapter does
# NOT run uncapped — it sends maxOutputTokens=65,535
# (GEMINI_DEFAULT_MAX_OUTPUT_TOKENS, see
# _effective_gemini_max_output_tokens). The compressor's threshold is
# pct×(window − max_tokens); passing None here meant it reserved 0
# while the wire reserved 65,535, so on a 128K window the trigger
# landed at ~96K against a real safe input budget of ~65K and the
# provider 400'd before compaction fired. Mirror the adapter's
# default so the reservation matches what is actually sent. The
# generic provider-default gap is #63839; this wires only the native
# Gemini path, where the default is a documented constant.
_compressor_max_tokens = agent.max_tokens
if _compressor_max_tokens is None:
try:
from agent.gemini_native_adapter import (
GEMINI_DEFAULT_MAX_OUTPUT_TOKENS,
is_native_gemini_base_url,
)
_gemini_provider = str(
getattr(agent, "provider", "") or ""
).strip().lower() in {
"gemini", "google", "google-gemini", "google-ai-studio",
}
if _gemini_provider or is_native_gemini_base_url(agent.base_url):
_compressor_max_tokens = GEMINI_DEFAULT_MAX_OUTPUT_TOKENS
except Exception:
pass
agent.context_compressor = ContextCompressor(
model=agent.model,
threshold_percent=compression_threshold,
@@ -2791,7 +2831,7 @@ def init_agent(
provider=agent.provider,
api_mode=agent.api_mode,
abort_on_summary_failure=compression_abort_on_summary_failure,
max_tokens=agent.max_tokens,
max_tokens=_compressor_max_tokens,
model_thresholds=compression_model_thresholds,
threshold_tokens_cap=compression_threshold_tokens,
proactive_prune_tokens=compression_proactive_prune_tokens,
@@ -2971,6 +3011,7 @@ def init_agent(
# until the first response with usage; invalidated on compaction and
# session switches so stale anchors can never suppress compression.
agent._usage_anchor = None
agent._turn_base_usage_anchor = None
# Cumulative token usage for the session
agent.session_prompt_tokens = 0
@@ -3040,6 +3081,36 @@ def init_agent(
"Ollama num_ctx: will request %d tokens (model max from /api/show)",
agent._ollama_num_ctx,
)
# ── Recalibrate the compressor to the served window (#57275 claim 3) ──
# The compressor was constructed ABOVE this block from the probed model
# window (GGUF metadata can advertise 256K+), but every request below
# runs at num_ctx. A config that sets only model.ollama_num_ctx (without
# model.context_length) previously left the compressor targeting the
# probed window while the server truncated/rejected at num_ctx — the
# compaction trigger could sit several times ABOVE the real served
# window and never fire. Clamp the compressor's window to the effective
# num_ctx so threshold math operates on the context the server actually
# serves. (Overlaps #60103's silent-clamp dead zone; this is the
# init-order half.)
_cc_window = getattr(agent.context_compressor, "context_length", 0) or 0
if (
agent._ollama_num_ctx
and agent._ollama_num_ctx > 0
and _cc_window
and agent._ollama_num_ctx < _cc_window
):
_ra().logger.info(
"Compressor window clamped to Ollama num_ctx: %d -> %d",
_cc_window, agent._ollama_num_ctx,
)
agent.context_compressor.update_model(
model=agent.model,
context_length=agent._ollama_num_ctx,
base_url=agent.base_url,
api_key=getattr(agent, "api_key", ""),
provider=agent.provider,
api_mode=agent.api_mode,
)
# Codex gpt-5.x autoraise notice: show at most once per profile/config
# state. Without the persisted marker the notice re-fires on every agent
+244 -10
View File
@@ -2842,9 +2842,9 @@ def create_openai_client(agent, client_kwargs: dict, *, reason: str, shared: boo
# All primary construction and recovery paths must identify Hermes to the
# official Codex endpoint, including snapshots with custom header overrides.
from agent.auxiliary_client import _apply_required_codex_headers
from agent.codex_headers import apply_required_codex_headers
_apply_required_codex_headers(
apply_required_codex_headers(
client_kwargs,
access_token=client_kwargs.get("api_key", ""),
base_url=str(client_kwargs.get("base_url", "")),
@@ -3856,6 +3856,25 @@ def _tool_call_id_variants(tc: Any) -> set:
# consistently whether the empty turn was caught at write time or send time.
_INTERRUPTED_PLACEHOLDER = "[response interrupted]"
# Repeated heals of the same poisoned transcript used to WARNING on every
# send (#96870). Escalate once per session window, then stay quiet.
# ``_EMPTY_HEAL_ESCALATE_AFTER`` is the built-in default; deployments tune it
# via ``agent.sanitizer_heal_escalation_threshold`` in config.yaml (<= 0
# disables escalation entirely — WARNINGs still fire per window).
_EMPTY_HEAL_ESCALATE_AFTER = 3
_EMPTY_HEAL_WINDOW_S = 600.0
_empty_heal_log_state: Dict[str, Dict[str, Any]] = {}
_empty_heal_log_lock = threading.Lock()
# Session keys that already received the one-time user notice. Separate from
# the windowed log state so a new 10-minute window never re-notifies: the
# user is told ONCE per session, ever (#96870 — out-of-band, delivery
# channel only, never injected into conversation context).
_empty_heal_user_notified: set = set()
# One-shot pending notices keyed by session, drained by the conversation
# loop through ``consume_pending_sanitizer_heal_notice`` and delivered via
# the status/warning callback (the normal delivery channel).
_empty_heal_pending_notice: Dict[str, str] = {}
def _msg_has_payload(msg: Dict[str, Any]) -> bool:
"""True if ``msg`` carries anything the API treats as non-empty content.
@@ -3905,6 +3924,167 @@ def _msg_has_payload(msg: Dict[str, Any]) -> bool:
return False
def fill_empty_non_final_wire_payload(
msg: Dict[str, Any], *, is_final: bool
) -> bool:
"""Write the interrupted placeholder onto an empty non-final wire copy.
Used by the send-time projection so ``repair_empty_non_final_messages``
does not re-heal the same row on every call (#88955 hidden placeholders,
#96870 stream-death / host-fed empties). Pass the per-call copy only —
durable history must not be mutated. Returns True when *msg* was filled.
"""
if is_final or not isinstance(msg, dict):
return False
if msg.get("role") not in ("user", "assistant"):
return False
if _msg_has_payload(msg):
return False
msg["content"] = _INTERRUPTED_PLACEHOLDER
return True
def _session_id_for_heal_log() -> str:
try:
from hermes_logging import _session_context
return str(getattr(_session_context, "session_id", None) or "")
except Exception:
return ""
def _heal_escalation_threshold() -> int:
"""Resolve the escalation threshold: config override, else the default.
``agent.sanitizer_heal_escalation_threshold`` in config.yaml. Fail-safe:
any read error falls back to the module default so the sanitiser can
never be broken by a bad config file.
"""
try:
from hermes_cli.config import load_config_readonly
raw = (load_config_readonly().get("agent", {}) or {}).get(
"sanitizer_heal_escalation_threshold"
)
if raw is not None:
return int(raw)
except Exception:
pass
return _EMPTY_HEAL_ESCALATE_AFTER
def consume_pending_sanitizer_heal_notice() -> Optional[str]:
"""Drain the one-time user notice for the current session, if any.
Called by the conversation loop right after the pre-send sanitizer pass;
the returned text is delivered through the status/warning callback (the
normal out-of-band delivery channel: gateway status message, CLI stderr
print). It is NEVER appended to the conversation context, so prompt
caching and role alternation are untouched. Returns at most one notice
per session for its whole lifetime.
"""
key = _session_id_for_heal_log() or "-"
with _empty_heal_log_lock:
return _empty_heal_pending_notice.pop(key, None)
def get_sanitizer_heal_stats() -> Dict[str, Dict[str, Any]]:
"""Read-only snapshot of per-session sanitiser heal counters.
Surfaced by diagnostics (``hermes doctor`` / debug share callers) so
repeated silent repairs are visible outside errors.log. Keys are session
ids; values carry ``heal_events`` (sanitizer invocations that healed at
least one message), ``messages_healed`` (total substituted turns) and
``escalated`` (whether the ERROR + user notice fired).
"""
with _empty_heal_log_lock:
return {
k: {
"heal_events": v.get("total_events", v.get("count", 0)),
"messages_healed": v.get("total_healed", 0),
"escalated": k in _empty_heal_user_notified,
}
for k, v in _empty_heal_log_state.items()
}
def _log_empty_non_final_heal(healed: int) -> None:
"""WARNING on the first heals in a window; one ERROR at the threshold.
Further heals in the same session window stay silent so a poisoned
transcript cannot flood ``errors.log`` (dozens of identical WARNINGs
per hour with no user-visible signal — #96870). At the threshold the
escalation also queues a ONE-TIME out-of-band user notice (drained by
``consume_pending_sanitizer_heal_notice``) pointing at ``/debug share``
/ ``hermes doctor`` — once per session, never re-armed by a new window.
"""
key = _session_id_for_heal_log() or "-"
threshold = _heal_escalation_threshold()
now = time.monotonic()
with _empty_heal_log_lock:
state = _empty_heal_log_state.get(key)
if state is None or (now - state["window_start"]) > _EMPTY_HEAL_WINDOW_S:
prior_events = state.get("total_events", 0) if state else 0
prior_healed = state.get("total_healed", 0) if state else 0
state = {
"count": 0,
"window_start": now,
"escalated": False,
"total_events": prior_events,
"total_healed": prior_healed,
}
_empty_heal_log_state[key] = state
state["count"] += 1
state["total_events"] = state.get("total_events", 0) + 1
state["total_healed"] = state.get("total_healed", 0) + healed
count = state["count"]
total_events = state["total_events"]
total_healed = state["total_healed"]
if threshold > 0 and count >= threshold and not state["escalated"]:
state["escalated"] = True
level = "error"
if key not in _empty_heal_user_notified:
_empty_heal_user_notified.add(key)
_empty_heal_pending_notice[key] = (
"⚠️ Your session transcript required repeated repair "
f"({total_events} heal passes so far). Replies keep "
"working, but a corrupted turn is stuck in this "
"session's history — run /debug share or `hermes "
"doctor` to capture diagnostics, or /new to start a "
"clean session."
)
elif state["escalated"]:
level = "silent"
else:
level = "warning"
if level == "silent":
return
if level == "error":
_ra().logger.error(
"Pre-call sanitizer: repeated-heal escalation for session %s — "
"healed %d empty non-final message(s) this send; heal pattern: "
"%d heal events / %d messages healed this session "
"(%d in the current session window, threshold %d). The transcript "
"is being repaired on every send; /new drops the poisoned turns.",
key,
healed,
total_events,
total_healed,
count,
threshold,
)
return
_ra().logger.warning(
"Pre-call sanitizer: healed %d empty non-final message(s) by "
"substituting placeholder content — an empty-content turn was in "
"the transcript and would 400 the request ('messages must have "
"non-empty content' / INVALID_REQUEST_BODY). Self-recovering the "
"poisoned transcript in memory; no restart needed.",
healed,
)
def repair_empty_non_final_messages(
messages: List[Dict[str, Any]],
) -> List[Dict[str, Any]]:
@@ -3961,14 +4141,7 @@ def repair_empty_non_final_messages(
repaired.append(msg)
if healed:
_ra().logger.warning(
"Pre-call sanitizer: healed %d empty non-final message(s) by "
"substituting placeholder content — an empty-content turn was in "
"the transcript and would 400 the request ('messages must have "
"non-empty content' / INVALID_REQUEST_BODY). Self-recovering the "
"poisoned transcript in memory; no restart needed.",
healed,
)
_log_empty_non_final_heal(healed)
return repaired
return messages
@@ -4373,6 +4546,67 @@ def sanitize_api_messages(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]
"Pre-call sanitizer: removed %d duplicate tool_call_id reference(s)",
removed_dupes,
)
# 4. Align each tool result's wire-visible ``name`` with the function name
# of the call it answers. Google matches functionResponse.name against
# functionCall.name and rejects a mismatch with HTTP 400 "Request contains
# an invalid argument" (INVALID_ARGUMENT); behind an OpenAI-compatible
# gateway that surfaces only as a generic "Provider returned error".
#
# The mismatch is routine, not corruption. When tool_search defers
# MCP/plugin tools the model calls the bridge tool ``tool_call``, while
# ``make_tool_result_message()`` labels the result with the unwrapped
# internal tool name (``mcp__github__create_issue``) that dispatch, hooks,
# logging, and guardrails need. #72089 fixed exactly this for the native
# Gemini adapter, which now prefers ``tool_name_by_call_id`` over the
# result name; requests that reach Gemini through the OpenAI-compatible
# path (OpenRouter, Vertex/LiteLLM proxies, any OpenAI-shaped gateway) skip
# that translation entirely and still send the internal name on the wire.
#
# Normalizing here rather than in the OpenAI-compat serializer keeps it
# provider-agnostic: Gemini reaches Hermes under many model strings and
# base URLs, so sniffing for "is this really Google?" is unreliable, and
# every other provider either ignores the field or agrees with the call
# name. Runs on the per-call copy, so the stored trajectory keeps the real
# tool name for the session DB and the UI — only the wire payload changes.
# A no-op for the native Gemini path, which already resolves the same name.
# A result whose assistant call frame is missing entirely never reaches
# here — pass 1 above drops it as an orphan — so the only results this pass
# sees are ones whose call name is knowable.
call_names: Dict[str, str] = {}
for msg in messages:
if msg.get("role") == "assistant":
for tc in msg.get("tool_calls") or []:
# Strip on insert to match the lookup below (and pass 1's
# ``result_call_ids``), so an id that arrives padded still
# pairs instead of silently skipping realignment.
cid = (_ra().AIAgent._get_tool_call_id_static(tc) or "").strip()
nm = _ra().AIAgent._get_tool_call_name_static(tc)
if cid and nm:
call_names[cid] = nm
realigned: List[Tuple[str, str]] = []
aligned: List[Dict[str, Any]] = []
for msg in messages:
if msg.get("role") == "tool":
cid = (msg.get("tool_call_id") or "").strip()
expected = call_names.get(cid)
current = msg.get("name")
# Only rewrite a name that is present and disagrees. A result with
# no ``name`` is already valid for Gemini (the id pairs it), so
# leave it absent rather than inventing a field: clean transcripts
# must still pass through byte-identical for prompt caching.
if expected and current and current != expected:
msg = {**msg, "name": expected}
realigned.append((current, expected))
aligned.append(msg)
if realigned:
messages = aligned
_ra().logger.debug(
"Pre-call sanitizer: realigned %d tool result name(s) with their "
"tool_call function name (%s)",
len(realigned),
", ".join(f"{was} -> {now}" for was, now in realigned),
)
return messages
+89 -3
View File
@@ -15,6 +15,7 @@ import json
import logging
import os
import platform
import re
import secrets
import stat
import subprocess
@@ -516,6 +517,55 @@ def _detect_claude_code_version() -> str:
_CLAUDE_CODE_SYSTEM_PREFIX = "You are Claude Code, Anthropic's official CLI for Claude."
_MCP_TOOL_PREFIX = "mcp__"
# Anthropic's OAuth billing classifier fingerprints certain Hermes tool
# schemas/prose as a third-party app and reroutes the request to the metered
# extra-usage lane, surfacing as HTTP 400 "You're out of extra usage" on a
# valid subscription token (#65365). Deterministic live A/B repros (issue
# #65365 comments, replayed with the anthropic-ratelimit-unified-* response
# headers as a lane oracle) isolated two independent triggers:
# - the ``session_search`` tool schema/name/prose, alone
# - the ``memory`` tool schema/name, alone
# Both are aliased to neutral names on the OAuth wire only. normalize_response
# reverses the mapping before dispatch, so tool behavior and API-key requests
# are unchanged.
_OAUTH_TOOL_NAME_ALIASES = {
"session_search": "chat_history_lookup",
"memory": "context_notes",
}
_OAUTH_TOOL_NAME_REVERSE_ALIASES = {
wire_name: name for name, wire_name in _OAUTH_TOOL_NAME_ALIASES.items()
}
# Aliases that are ALSO safe to substitute in free-form prose (system prompt
# text, tool descriptions). Only unambiguous snake_case tool tokens qualify:
# "memory" is ordinary English throughout the system prompt ("persistent
# memory across sessions", "OS, CPU, memory, disk") and inside the memory
# tool's own parameter docs (the ``target`` enum the model must still emit
# verbatim), so rewriting it in prose would corrupt guidance the model has
# to follow. Renaming a tool is a different operation from rewriting the
# vocabulary that describes it — a model that follows unaliased "memory"
# prose and calls ``memory`` still dispatches correctly: normalize_response
# resolves the bare name through the tool registry regardless.
_OAUTH_PROSE_ALIAS_NAMES = frozenset({"session_search"})
# Word-boundary matchers so a prose substitution can't corrupt a longer
# identifier that merely contains the token (project AGENTS.md / memory
# snapshots can carry arbitrary text, e.g. a path like
# ``tools/session_search_tool.py`` must not become
# ``tools/chat_history_lookup_tool.py``). ``\b`` treats ``_`` as a word
# char, so only the standalone token matches.
_OAUTH_PROSE_ALIAS_PATTERNS = tuple(
(re.compile(rf"\b{re.escape(name)}\b"), _OAUTH_TOOL_NAME_ALIASES[name])
for name in sorted(_OAUTH_PROSE_ALIAS_NAMES)
)
def _apply_oauth_prose_aliases(text: str) -> str:
"""Rewrite prose-safe tool-name tokens to their OAuth wire aliases."""
for pattern, wire_name in _OAUTH_PROSE_ALIAS_PATTERNS:
text = pattern.sub(wire_name, text)
return text
def _get_claude_code_version() -> str:
"""Lazily detect the installed Claude Code version when OAuth headers need it."""
@@ -936,6 +986,7 @@ def build_anthropic_kwargs(
text = text.replace("Hermes agent", "Claude Code")
text = text.replace("hermes-agent", "claude-code")
text = text.replace("Nous Research", "Anthropic")
text = _apply_oauth_prose_aliases(text)
block["text"] = text
# 3. Normalize tool names so NOTHING goes on the OAuth wire with a
@@ -956,7 +1007,14 @@ def build_anthropic_kwargs(
# so any session with an MCP server configured still tripped the
# classifier. normalize_response reverses both forms via registry
# lookup so the dispatcher still sees the original name. GH-25255.
def _to_oauth_wire_name(name: str) -> str:
# Wire names owned by tools that are NOT alias sources. An alias must
# never collide with one: two identical tool names in a single
# request is a hard 400 from Anthropic, strictly worse than the bug
# being fixed. Mirrors the "registered tool wins" precedence in
# normalize_response so outbound and inbound agree on who owns a
# contested name.
def _normalize_to_mcp_wire(name: str) -> str:
"""OAuth wire form of a tool name (no aliasing): mcp__<...>."""
if name.startswith("mcp__"):
return name # already correct, don't double-prefix
if name.startswith("mcp_"):
@@ -964,10 +1022,28 @@ def build_anthropic_kwargs(
return "mcp__" + name[len("mcp_"):]
return _MCP_TOOL_PREFIX + name # bare name -> mcp__<name>
_claimed_wire_names = {
_normalize_to_mcp_wire(tool["name"])
for tool in (anthropic_tools or [])
if isinstance(tool.get("name"), str)
and tool["name"] not in _OAUTH_TOOL_NAME_ALIASES
}
def _to_oauth_wire_name(name: str) -> str:
if name in _OAUTH_TOOL_NAME_ALIASES:
aliased = _OAUTH_TOOL_NAME_ALIASES[name]
if _MCP_TOOL_PREFIX + aliased not in _claimed_wire_names:
name = aliased
return _normalize_to_mcp_wire(name)
if anthropic_tools:
for tool in anthropic_tools:
if "name" in tool:
tool["name"] = _to_oauth_wire_name(tool["name"])
description = tool.get("description")
if isinstance(description, str):
# Prose-safe aliases only — see _OAUTH_PROSE_ALIAS_NAMES.
tool["description"] = _apply_oauth_prose_aliases(description)
# 4. Apply the same normalization to tool names in message history
# (tool_use blocks) so replayed turns match the wire names above.
@@ -1001,8 +1077,18 @@ def build_anthropic_kwargs(
# Anthropic has no tool_choice "none" — omit tools entirely to prevent use
kwargs.pop("tools", None)
elif isinstance(tool_choice, str):
# Specific tool name
kwargs["tool_choice"] = {"type": "tool", "name": tool_choice}
# Specific tool name. Under OAuth every tool on the wire is
# mcp__-prefixed and/or alias-renamed (see _to_oauth_wire_name
# above) — route the forced name through the same normalizer so
# tool_choice always matches the corresponding tools[] entry.
# Left un-normalized, a forced ``session_search``/``memory``
# choice would (a) still carry the literal trigger string onto
# the wire, defeating the alias, and (b) reference a tool name
# that no longer exists in ``tools[]``, which Anthropic rejects.
wire_tool_choice = tool_choice
if is_oauth:
wire_tool_choice = _to_oauth_wire_name(tool_choice)
kwargs["tool_choice"] = {"type": "tool", "name": wire_tool_choice}
# Map reasoning_config to Anthropic's thinking parameter.
# Claude 4.6+ models use adaptive thinking + output_config.effort.
+566 -123
View File
@@ -62,6 +62,13 @@ from types import SimpleNamespace
from typing import Any, Callable, Dict, List, NamedTuple, Optional, Tuple, TYPE_CHECKING
from urllib.parse import urlparse, parse_qs, urlunparse
from agent.codex_headers import (
CODEX_AUX_BASE_URL as _CODEX_AUX_BASE_URL,
apply_required_codex_headers as _apply_required_codex_headers,
codex_cloudflare_headers as _codex_cloudflare_headers,
is_official_codex_base_url as _is_official_codex_base_url,
)
# NOTE: `from openai import OpenAI` is deliberately NOT at module top — the
# openai SDK pulls a large type tree (~240 ms cold, including responses/*,
# graders/*). We expose `OpenAI` here as a thin proxy that imports the SDK on
@@ -530,6 +537,14 @@ _CODEX_PROGRESS_DELTA_TYPES = frozenset(
)
# Progress-aware auxiliary stream deadlines (Aug 2026, masoria report):
# a dead stream fails fast at the no-progress window (first token AND
# between tokens), a live stream re-arms per substantive event and is
# bounded only by _aux_stream_total_ceiling() (shared with the streamed
# chat.completions path).
_AUX_STREAM_NO_PROGRESS_TIMEOUT_SECONDS = 60.0
def _codex_event_has_content(event: Any) -> bool:
"""Whether a Codex Responses event carries a non-empty payload."""
event_type = _event_field(event, "type")
@@ -978,6 +993,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
network path — the underlying fetch is memory+disk cached with a
last-known-good fallback.
"""
is_nous = provider_id.strip().lower() == "nous"
try:
from hermes_cli.auth import resolve_api_key_provider_credentials
from hermes_cli.models import fetch_models_with_pricing
@@ -997,6 +1013,17 @@ def _fast_model_from_catalog(provider_id: str) -> str:
# fetch below still works for the catalogs that allow it.
logger.debug("No credentials for %s catalog", provider_id, exc_info=True)
if not api_key and is_nous:
# Nous is OAuth, so the resolver above raises for it. An anonymous
# read returns the full catalog, and a model picked from it is
# refused at request time by the org's policy.
try:
from hermes_cli.models import _resolve_nous_pricing_credentials
api_key, base_url = _resolve_nous_pricing_credentials()
except Exception:
logger.debug("No Nous credentials for catalog", exc_info=True)
if not base_url:
base_url = str(getattr(get_provider_profile(provider_id), "base_url", "") or "")
base_url = base_url.rstrip("/")
@@ -1005,14 +1032,37 @@ def _fast_model_from_catalog(provider_id: str) -> str:
# fetch_models_with_pricing appends its own /v1/models.
if base_url.endswith("/v1"):
base_url = base_url[:-3]
# Same entry the pickers use, so the Nous-only arguments must match
# theirs: seeding it here without them costs the picker its sale chrome
# and leaves the policy catalog with no expiry.
_nous_kwargs = {}
if is_nous:
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
_nous_kwargs = {
"include_sale_original": True,
"cache_ttl_seconds": _NOUS_CATALOG_TTL_SECONDS,
}
catalog = fetch_models_with_pricing(
api_key=api_key or None, base_url=base_url, timeout=3.0
api_key=api_key or None, base_url=base_url, timeout=3.0, **_nous_kwargs
) or {}
except Exception:
logger.debug("Fast-model catalog lookup failed for %s", provider_id, exc_info=True)
return ""
ids = sorted((str(m) for m in catalog), key=_model_recency_key, reverse=True)
if is_nous:
# The catalog's keys are a source of ids here, so the policy narrows
# them as it does the pickers' lists.
try:
from hermes_cli.models import (
nous_policy_allowed_ids,
restrict_to_nous_policy,
)
ids = restrict_to_nous_policy(ids, nous_policy_allowed_ids())
except Exception:
logger.debug("Nous policy filter unavailable", exc_info=True)
for family in _FAST_MODEL_FAMILIES:
for model_id in ids:
lowered = model_id.lower()
@@ -1021,6 +1071,18 @@ def _fast_model_from_catalog(provider_id: str) -> str:
return ""
def _nous_policy_blocks(model_id: str) -> bool:
"""True when the org's model policy does not admit *model_id*."""
try:
from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
allowed = nous_policy_allowed_ids()
return bool(allowed) and not restrict_to_nous_policy([model_id], allowed)
except Exception:
logger.debug("Nous policy check unavailable", exc_info=True)
return False
# Default auxiliary models for direct API-key providers (cheap/fast for side tasks)
def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False) -> str:
"""Return the cheap auxiliary model for a provider.
@@ -1048,21 +1110,26 @@ def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False)
except Exception:
pass
picked = ""
if prefer_fast:
catalog_pick = _fast_model_from_catalog(provider_id)
if catalog_pick:
return catalog_pick
if profile is not None:
picked = _fast_model_from_catalog(provider_id)
if not picked and profile is not None:
try:
live = profile.resolve_aux_model()
if live:
return live
picked = profile.resolve_aux_model() or ""
except Exception:
logger.debug("resolve_aux_model failed for %s", provider_id, exc_info=True)
if profile is not None and profile.default_aux_model:
return profile.default_aux_model
return _API_KEY_PROVIDER_AUX_MODELS_FALLBACK.get(provider_id, "")
if not picked and profile is not None and profile.default_aux_model:
picked = profile.default_aux_model
if not picked:
picked = _API_KEY_PROVIDER_AUX_MODELS_FALLBACK.get(provider_id, "")
# Steps 2-4 are policy-blind: resolve_aux_model queries a public
# recommendation and the rest are hardcoded. A blocked pick is refused at
# request time, so drop it and let the caller keep the main model.
if picked and provider_id.strip().lower() == "nous" and _nous_policy_blocks(picked):
return ""
return picked
@@ -1319,96 +1386,24 @@ NOUS_EXTRA_BODY = _nous_extra_body()
# Set at resolve time — True if the auxiliary client points to Nous Portal
auxiliary_is_nous: bool = False
# Default auxiliary models per provider
_OPENROUTER_MODEL = "google/gemini-3.6-flash"
# Default auxiliary models per provider.
# _OPENROUTER_MODEL is the BUILT-IN fallback used only when the user never set
# auxiliary.openrouter_model — it MUST be a :free SKU: this lane engages
# silently (auxiliary tasks, no user prompt), and a paid built-in default here
# meant silent real OpenRouter spend the user never opted into (#81952). The
# SKU matches the one the free_only warning below recommends. User-configured
# auxiliary.openrouter_model values are honored untouched (paid allowed when
# the user chose it; _warn_paid_lane_once still fires for that case).
_OPENROUTER_MODEL = "nvidia/nemotron-3-ultra-550b-a55b:free"
_NOUS_MODEL = "google/gemini-3.6-flash"
_NOUS_DEFAULT_BASE_URL = "https://inference-api.nousresearch.com/v1"
_ANTHROPIC_DEFAULT_BASE_URL = "https://api.anthropic.com"
_AUTH_JSON_PATH = get_hermes_home() / "auth.json"
# Codex OAuth endpoint used when a caller explicitly requests
# provider="openai-codex". There is deliberately no hardcoded default
# model: the set of models OpenAI accepts on this endpoint for
# ChatGPT-account auth is an undocumented, shifting allow-list, and
# pinning one here has drifted silently twice (gpt-5.3-codex → gpt-5.2-codex
# → gpt-5.4 over 6 weeks in early 2026). Callers must pass the model
# they want explicitly (from config.yaml model.model, auxiliary.<task>.model,
# or the user's active Codex model selection).
_CODEX_AUX_BASE_URL = "https://chatgpt.com/backend-api/codex"
def _is_official_codex_base_url(base_url: str) -> bool:
"""Identify OpenAI's Codex endpoint without matching custom proxies."""
try:
parsed = urlparse(base_url)
path = parsed.path.rstrip("/")
return (
parsed.scheme == "https"
and parsed.hostname == "chatgpt.com"
and parsed.port in (None, 443)
and (path == "/backend-api/codex" or path.startswith("/backend-api/codex/"))
)
except (TypeError, ValueError):
return False
def _codex_cloudflare_headers(
access_token: str, *, base_url: str = _CODEX_AUX_BASE_URL,
) -> Dict[str, str]:
"""Identity and account headers for chatgpt.com/backend-api/codex.
OpenAI requires third-party harnesses to identify themselves. Requests to
the official endpoint always send Hermes' originator and version. Custom
endpoints retain the existing compatibility identity. In either case,
preserve ``ChatGPT-Account-ID`` from the OAuth JWT's
``chatgpt_account_id`` claim.
Malformed tokens are tolerated — we drop the account-ID header rather than
raise, so a bad token still surfaces as an auth error (401) instead of a
crash at client construction.
"""
headers = {
"User-Agent": "codex_cli_rs/0.0.0 (Hermes Agent)",
"originator": "codex_cli_rs",
}
if _is_official_codex_base_url(base_url):
from hermes_cli import __version__
headers.update({
"User-Agent": f"HermesAgent/{__version__}",
"originator": "hermes-agent",
})
if not isinstance(access_token, str) or not access_token.strip():
return headers
try:
import base64
parts = access_token.split(".")
if len(parts) < 2:
return headers
payload_b64 = parts[1] + "=" * (-len(parts[1]) % 4)
claims = json.loads(base64.urlsafe_b64decode(payload_b64))
acct_id = claims.get("https://api.openai.com/auth", {}).get("chatgpt_account_id")
if isinstance(acct_id, str) and acct_id:
headers["ChatGPT-Account-ID"] = acct_id
except Exception:
pass
return headers
def _apply_required_codex_headers(
client_kwargs: Dict[str, Any], *, access_token: str, base_url: str,
) -> None:
"""Keep required Codex identity after user/provider header overrides."""
if not _is_official_codex_base_url(base_url):
return
required = _codex_cloudflare_headers(access_token, base_url=base_url)
required_names = {name.lower() for name in required}
existing = client_kwargs.get("default_headers") or {}
client_kwargs["default_headers"] = {
**{name: value for name, value in existing.items()
if str(name).lower() not in required_names},
**required,
}
# Codex helpers live in a small leaf module so fresh client builders never
# request newly added exports from a stale, long-lived auxiliary router. The
# private aliases above preserve the established import surface for plugins and
# tests while new production consumers import the leaf directly.
# Hosts that expose BOTH an Anthropic-style ``…/anthropic`` path and a sibling
@@ -1720,6 +1715,25 @@ class _CodexCompletionsAdapter:
# same behavior as the main agent's Codex transport.
extra_body = kwargs.get("extra_body") or {}
if isinstance(extra_body, dict):
# Fast mode / Priority Processing is a top-level Responses field.
# Auxiliary callers express provider controls through
# auxiliary.<task>.extra_body, so project service_tier here just as
# the main Codex transport projects request_overrides. xAI's
# Responses endpoint rejects this field; keep the same xAI-only
# guard as agent/transports/codex.py.
service_tier = extra_body.get("service_tier")
client_base_url = str(getattr(self._client, "base_url", "") or "")
is_xai_responses = (
base_url_host_matches(client_base_url, "x.ai")
or base_url_host_matches(client_base_url, "api.x.ai")
)
if (
isinstance(service_tier, str)
and service_tier.strip()
and not is_xai_responses
):
resp_kwargs["service_tier"] = service_tier.strip()
reasoning_cfg = extra_body.get("reasoning")
if isinstance(reasoning_cfg, dict):
if reasoning_cfg.get("enabled") is False:
@@ -1849,9 +1863,42 @@ class _CodexCompletionsAdapter:
tool_calls_raw: List[Any] = []
usage = None
total_timeout = timeout if isinstance(timeout, (int, float)) and timeout > 0 else None
deadline = time.monotonic() + float(total_timeout) if total_timeout else None
# Progress-aware stream deadlines (supersedes the old single absolute
# kill at ``total_timeout``). Three regimes:
# 1. First token: the stream must produce its first substantive
# payload within ``no_progress_timeout`` (60s default) or we
# fail fast and let the caller's normal retry/fallback chain
# run — a dead (or keepalive-only zombie) Codex stream no
# longer holds the full 300s compression budget before falling
# back (masoria report, Aug 2026: 3 stacked 300s waits ->
# 20+ min stuck on "Summarizing").
# 2. Streaming: every substantive event re-arms the deadline by
# ``no_progress_timeout`` — a live stream is never killed by an
# absolute total, so a long reasoning summary that is actually
# producing tokens completes instead of timing out at 300s and
# falling back (#54915's original complaint, fixed properly).
# Keepalive/lifecycle frames do NOT re-arm, mirroring the
# commit-fence progress gating (#96707).
# 3. Hard ceiling: an absolute backstop from
# ``_aux_stream_total_ceiling`` (max(600s, 4x configured
# timeout) — the same bound the streamed chat.completions path
# uses) so a pathological one-token-per-59s drip still
# terminates.
_start_monotonic = time.monotonic()
no_progress_timeout = _AUX_STREAM_NO_PROGRESS_TIMEOUT_SECONDS
if total_timeout is not None:
no_progress_timeout = min(no_progress_timeout, float(total_timeout))
hard_deadline = _start_monotonic + _aux_stream_total_ceiling(total_timeout)
deadline_lock = threading.Lock()
progress_deadline = [_start_monotonic + no_progress_timeout]
saw_content = threading.Event()
timed_out = threading.Event()
timeout_timer: Optional[threading.Timer] = None
# Set only when the timeout WON the attempt (not when the owner
# hard-cancelled first): tells the owning thread's ``finally`` that
# the shared client's FDs still need a real close (#29507).
timeout_release_pending = threading.Event()
stream_finished = threading.Event()
timeout_timer: List[Optional[threading.Timer]] = [None]
# A protected provider call may outlive its owning compression attempt:
# the owner returns promptly on hard cancellation while this adapter is
# still blocked in the SDK stream on its isolated worker. Timer threads
@@ -1862,9 +1909,38 @@ class _CodexCompletionsAdapter:
)
attempt_stream_lock = threading.Lock()
attempt_stream: List[Any] = []
# The thread driving this request owns its transport's file
# descriptors — see the FD-ownership note in _close_client_on_timeout.
owner_tid = threading.get_ident()
def _effective_deadline() -> float:
with deadline_lock:
return min(hard_deadline, progress_deadline[0])
def _record_stream_progress() -> None:
# A substantive payload re-arms the no-progress window. The hard
# ceiling is never extended.
with deadline_lock:
progress_deadline[0] = time.monotonic() + no_progress_timeout
def _timeout_message() -> str:
return f"Codex auxiliary Responses stream exceeded {float(total_timeout):.1f}s total timeout"
elapsed = time.monotonic() - _start_monotonic
if time.monotonic() >= hard_deadline:
return (
"Codex auxiliary Responses stream exceeded "
f"{hard_deadline - _start_monotonic:.1f}s hard ceiling"
)
if not saw_content.is_set():
return (
"Codex auxiliary Responses stream produced no output "
f"within {float(no_progress_timeout):.1f}s "
f"(no-progress timeout, {elapsed:.1f}s elapsed)"
)
return (
"Codex auxiliary Responses stream stalled: no new output "
f"for {float(no_progress_timeout):.1f}s "
f"({elapsed:.1f}s elapsed)"
)
def _close_client_on_timeout() -> None:
begin_timeout_cleanup = getattr(
@@ -1899,12 +1975,58 @@ class _CodexCompletionsAdapter:
exc_info=True,
)
return
close = getattr(self._client, "close", None)
if callable(close):
# FD-ownership contract (#29507 / #67142 / #70773): only the
# thread driving the request may RELEASE this client's file
# descriptors. This callback has two callers — ``_check_cancelled``
# on the owning thread, and the daemon watchdog ``threading.Timer``,
# which is a stranger thread. From a stranger thread we may only
# ``shutdown()`` the pooled sockets: ``close()`` releases the raw
# TLS fd while the owner's OpenSSL BIO still caches that integer,
# the kernel recycles it into the next ``open()`` in this process
# (a SessionDB / kanban.db handle), and the owner's unwinding TLS
# flush writes an application-data record into that database file.
# ``shutdown()`` from any thread is FD-safe; ``close()`` is not.
# The owning thread performs the real close in the ``finally``
# below, which is where the FD release belongs.
timeout_release_pending.set()
if threading.get_ident() == owner_tid:
close = getattr(self._client, "close", None)
if callable(close):
try:
close()
except Exception:
logger.debug("Codex auxiliary: client close during timeout failed", exc_info=True)
else:
try:
close()
from agent.agent_runtime_helpers import force_close_tcp_sockets
shutdown_count = force_close_tcp_sockets(self._client)
logger.info(
"Codex auxiliary client aborted (timeout, tcp_force_closed=%d, "
"deferred_close=stranger_thread)",
shutdown_count,
)
except Exception:
logger.debug("Codex auxiliary: client close during timeout failed", exc_info=True)
logger.debug("Codex auxiliary: client abort during timeout failed", exc_info=True)
# Socket shutdown wakes a reader blocked on a REAL transport,
# but the owner may be blocked inside the SDK's event stream
# (or a test double with no sockets). Closing the attempt-
# owned stream is the same attempt-scoped wake the hard-cancel
# branch above performs from this Timer thread — it releases
# the owner without touching the shared client's FDs; the
# owner then does the real close in its ``finally``.
with attempt_stream_lock:
stream = attempt_stream[0] if attempt_stream else None
close_stream = getattr(stream, "close", None)
if callable(close_stream):
try:
close_stream()
except Exception:
logger.debug(
"Codex auxiliary: attempt stream close during "
"stranger-thread timeout failed",
exc_info=True,
)
# The cached auxiliary client wraps this same ``self._client``
# (or *is* a ``CodexAuxiliaryClient`` whose ``_real_client`` is
# this instance). After we close the httpx transport above, the
@@ -1917,7 +2039,7 @@ class _CodexCompletionsAdapter:
logger.debug("Codex auxiliary: cache eviction on timeout failed", exc_info=True)
def _check_cancelled() -> None:
if deadline is not None and time.monotonic() >= deadline:
if total_timeout is not None and time.monotonic() >= _effective_deadline():
if not timed_out.is_set():
_close_client_on_timeout()
raise TimeoutError(_timeout_message())
@@ -1939,11 +2061,30 @@ class _CodexCompletionsAdapter:
# new failure mode for auxiliary calls.
pass
def _watchdog_fire() -> None:
# Re-armable watchdog: if progress moved the deadline forward
# since this timer was scheduled, reschedule instead of killing
# a live stream. Only kill when the effective deadline (progress
# window or hard ceiling, whichever is sooner) has truly passed.
remaining = _effective_deadline() - time.monotonic()
if remaining > 0:
if timed_out.is_set() or stream_finished.is_set():
return
t = threading.Timer(remaining, _watchdog_fire)
t.daemon = True
timeout_timer[0] = t
t.start()
return
_close_client_on_timeout()
try:
if total_timeout:
timeout_timer = threading.Timer(float(total_timeout), _close_client_on_timeout)
timeout_timer.daemon = True
timeout_timer.start()
timeout_timer[0] = threading.Timer(
max(_effective_deadline() - time.monotonic(), 0.0),
_watchdog_fire,
)
timeout_timer[0].daemon = True
timeout_timer[0].start()
_check_cancelled()
# Event-driven Responses streaming via the low-level
@@ -1975,7 +2116,13 @@ class _CodexCompletionsAdapter:
# compression commit fence) counts only substantive
# payloads — lifecycle and keepalive events must not reset
# the compression idle clock.
# The transport no-progress window likewise re-arms only on
# substantive payloads: a zombie stream that drips SSE
# keepalives but never produces output dies at the same 60s
# window as a fully dead connection.
if _codex_event_has_content(_event):
_record_stream_progress()
saw_content.set()
_notify_aux_provider_response()
else:
_notify_aux_timing_response()
@@ -2070,8 +2217,26 @@ class _CodexCompletionsAdapter:
logger.debug("Codex auxiliary Responses API call failed: %s", exc)
raise
finally:
if timeout_timer is not None:
timeout_timer.cancel()
stream_finished.set()
_t = timeout_timer[0]
if _t is not None:
_t.cancel()
# A stranger-thread timeout only shut the sockets down; the FDs
# are still open and this — the owning thread, now unwound — is
# the one context that may release them (#29507). Gated on
# timeout_release_pending, NOT timed_out: in the hard-cancel
# branch (timeout_won=False) the shared client must stay usable
# for other sessions.
if timeout_release_pending.is_set():
close = getattr(self._client, "close", None)
if callable(close):
try:
close()
except Exception:
logger.debug(
"Codex auxiliary: owner-thread close after timeout failed",
exc_info=True,
)
content = "".join(text_parts).strip() or None
@@ -6912,6 +7077,40 @@ def resolve_provider_client(
custom_key = build_command_token_provider(
custom_key_cmd, custom_entry.get("name") or provider
) or custom_key
if not custom_key:
try:
from agent.credential_pool import (
custom_provider_pool_key_candidates,
load_pool,
)
pool_name = (
custom_entry.get("provider_key")
or custom_entry.get("name")
or provider
)
for pool_key in custom_provider_pool_key_candidates(
custom_base, pool_name
):
try:
pool = load_pool(pool_key)
except Exception:
continue
if not pool.has_credentials():
continue
pool_entry = pool.select()
if pool_entry is None:
continue
pool_api_key = (
getattr(pool_entry, "runtime_api_key", None)
or getattr(pool_entry, "access_token", "")
or ""
)
if str(pool_api_key).strip():
custom_key = str(pool_api_key).strip()
break
except Exception:
pass
custom_key = custom_key or "no-key-required"
if custom_key == "no-key-required":
logger.warning(
@@ -9030,12 +9229,32 @@ def _build_call_kwargs(
from hermes_cli.providers import nous_api_mode
_nous_on_messages = nous_api_mode(model) == "anthropic_messages"
# OpenRouter budgets credit against the requested output cap; when the
# param is omitted it assumes the model's FULL output window (e.g.
# 65,536), so low-credit accounts 402 ("can only afford N") even
# though the actual summary would cost far less. Preserving the
# caller's cap keeps the request affordable (#41035, PR #41055 by
# @liuhao1024).
_is_openrouter = (
_provider_norm == "openrouter"
or base_url_host_matches(_effective_base, "openrouter.ai")
)
# The managed local llama-server honors explicit caps too: a local
# decode burns the user's own GPU at full tilt, so a caller that
# says "this is a 64-token task" must be believed — an uncapped
# local generation whose EOS never comes runs to the full context
# window. No wire-format quirks apply (llama.cpp accepts
# max_tokens), and the no-default-cap policy is unchanged: this
# only forwards caps callers explicitly set.
_is_managed_local = _is_managed_local_endpoint(_effective_base)
if (
_is_anthropic_compat_endpoint(provider, _effective_base)
or _nous_on_messages
or _is_nvidia_nim
or _is_moa
or _is_gemini_native
or _is_openrouter
or _is_managed_local
):
# Use auxiliary_max_tokens_param() so models that require
# max_completion_tokens (GPT-5 family, Copilot) get the right
@@ -9381,6 +9600,49 @@ def _is_streaming_rejected_error(exc: Exception) -> bool:
)
_MANAGED_LOCAL_STATE_TTL_S = 15.0
_managed_local_cache: "tuple[float, str]" = (0.0, "")
def _managed_local_netloc() -> str:
"""host:port of the managed local llama-server, or "" when none.
Read from the supervisor's state file (written at spawn, removed on
stop) with a short TTL so per-request checks don't hit the disk. The
state file is the same source provider resolution uses, so the match
is exact — no false positives on other localhost endpoints.
"""
global _managed_local_cache
now = time.monotonic()
ts, cached = _managed_local_cache
if now - ts < _MANAGED_LOCAL_STATE_TTL_S:
return cached
netloc = ""
try:
from hermes_cli.local_runtime.supervisor import state_path
raw = state_path().read_text(encoding="utf-8")
base = str((json.loads(raw) or {}).get("base_url", ""))
netloc = urlparse(base).netloc.lower()
except Exception:
netloc = ""
_managed_local_cache = (now, netloc)
return netloc
def _is_managed_local_endpoint(base_url: Optional[str]) -> bool:
"""True when *base_url* targets the llama-server this Hermes manages."""
if not base_url:
return False
managed = _managed_local_netloc()
if not managed:
return False
try:
return urlparse(str(base_url)).netloc.lower() == managed
except Exception:
return False
def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
"""Detect providers that only accept streaming (non-stream = HTTP 400).
@@ -9396,6 +9658,18 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
Beyond the known-host list, users can mark ANY custom endpoint as
stream-only via ``auxiliary.stream_only_base_urls`` in config.yaml
(list of substrings matched against the endpoint URL).
The managed local llama-server is always streamed for a different
reason: cancellation. llama-server only notices a dead client when it
writes to the socket. A non-streamed request writes once — after the
FULL generation — so an abandoned call (client timeout, retry, app
exit) keeps the GPU decoding to the end of the context window with
nobody listening; requests that queue behind a model load are the
worst case, since the client is long gone before decode even starts.
Streaming writes every few tokens, so an abandoned decode dies at the
first post-disconnect chunk (verified against llama-server b10362:
streamed disconnect cancels in <1s through the router; non-streamed
survives until the server's next incidental socket poll, if ever).
"""
_url = str(base_url or "").lower()
if not _url:
@@ -9403,6 +9677,9 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
# Tencent Copilot — "Non-stream chat request is currently not supported"
if base_url_host_matches(_url, "copilot.tencent.com"):
return True
# Managed local llama-server — streamed so abandonment cancels decode.
if _is_managed_local_endpoint(_url):
return True
try:
from hermes_cli.config import load_config
aux_cfg = (load_config() or {}).get("auxiliary", {})
@@ -9417,12 +9694,99 @@ def _provider_requires_stream(provider: str, base_url: Optional[str]) -> bool:
return False
_AFFORDABLE_TOKENS_RE = re.compile(
r"can only afford\s+([0-9][0-9,]*)", re.IGNORECASE
)
# Below this, the affordable budget can't fit a useful auxiliary output
# (summaries, titles, vision descriptions) — treat as genuine exhaustion.
_AFFORDABLE_RETRY_FLOOR_TOKENS = 512
# Headroom under the provider's stated budget so token-count rounding on
# their side can't 402 the retry (same margin PR #49785 used).
_AFFORDABLE_RETRY_MARGIN_TOKENS = 64
def _affordable_max_tokens_from_error(exc: Exception) -> Optional[int]:
"""Extract the affordable output budget from a credit-limited 402.
OpenRouter's insufficient-credit rejection states the budget explicitly:
``402 - This request requires more credits, or fewer max_tokens. You
requested up to 65536 tokens, but can only afford 7117.`` The account
HAS usable credit — the request just asked for (or defaulted to) an
output cap larger than the balance covers. Returns the retryable cap
(affordable minus a safety margin), or ``None`` when the error carries
no affordable count or the budget is too small to be useful.
"""
if not _is_payment_error(exc):
return None
match = _AFFORDABLE_TOKENS_RE.search(str(exc))
if not match:
return None
try:
affordable = int(match.group(1).replace(",", ""))
except (TypeError, ValueError):
return None
capped = affordable - _AFFORDABLE_RETRY_MARGIN_TOKENS
if capped < _AFFORDABLE_RETRY_FLOOR_TOKENS:
return None
return capped
def _create_with_progress(
client: Any,
kwargs: Dict[str, Any],
task: Optional[str] = None,
*,
force_stream: bool = False,
) -> Any:
"""Credit-aware wrapper over :func:`_create_with_progress_once`.
A 402 that names an affordable output budget ("can only afford N
tokens") is NOT terminal billing exhaustion — the account can pay for
the call at a lower ``max_tokens``. Retry ONCE with the provider-stated
cap (masoria debug bundle, Aug 2026: compression fell back to
OpenRouter, which defaulted the omitted cap to the model's full 65,536
window and 402'd three times in a row on an account that could afford
7,117 tokens — plenty for a summary). Only lowers, never raises, an
existing cap; anything else re-raises for the normal recovery chains.
"""
try:
return _create_with_progress_once(
client, kwargs, task, force_stream=force_stream,
)
except Exception as exc:
affordable = _affordable_max_tokens_from_error(exc)
if affordable is None:
raise
existing_cap = kwargs.get("max_tokens") or kwargs.get("max_completion_tokens")
if isinstance(existing_cap, (int, float)) and 0 < existing_cap <= affordable:
# The request was already within the stated budget — the error
# is something else (e.g. prompt-side cost). Don't spin.
raise
retry_kwargs = dict(kwargs)
retry_kwargs.pop("max_tokens", None)
retry_kwargs.pop("max_completion_tokens", None)
retry_kwargs.update(
auxiliary_max_tokens_param(
affordable, model=str(kwargs.get("model") or "") or None,
)
)
logger.info(
"Auxiliary %s: credit-limited 402 (affordable=%d tokens); "
"retrying once with a clamped output cap instead of failing: %s",
task or "call", affordable, exc,
)
return _create_with_progress_once(
client, retry_kwargs, task, force_stream=force_stream,
)
def _create_with_progress_once(
client: Any,
kwargs: Dict[str, Any],
task: Optional[str] = None,
*,
force_stream: bool = False,
) -> Any:
"""chat.completions.create() that streams when a progress hook is active
or the provider only accepts streamed requests.
@@ -9535,6 +9899,7 @@ class _ChatStreamAccumulator:
self._total_ceiling = total_ceiling
self.content_parts: List[str] = []
self.reasoning_parts: List[str] = []
self.reasoning_details: List[Any] = []
self.tool_calls_acc: Dict[int, Dict[str, Any]] = {}
self.finish_reason = None
self.usage = None
@@ -9579,6 +9944,24 @@ class _ChatStreamAccumulator:
if reasoning_piece and isinstance(reasoning_piece, str):
self.reasoning_parts.append(reasoning_piece)
made_progress = True
# OpenRouter-compatible reasoning models may stream their entire
# thinking phase through ``reasoning_details`` instead of
# ``reasoning`` / ``reasoning_content``. Treat only details carrying
# actual text as forward progress: structural/signed envelopes must
# not keep an otherwise stalled compression alive indefinitely.
reasoning_details = getattr(delta, "reasoning_details", None)
if reasoning_details is None:
model_extra = getattr(delta, "model_extra", None)
if isinstance(model_extra, dict):
reasoning_details = model_extra.get("reasoning_details")
if isinstance(reasoning_details, list):
for detail in reasoning_details:
self.reasoning_details.append(detail)
if isinstance(detail, dict) and any(
isinstance(detail.get(field), str) and detail[field]
for field in ("summary", "thinking", "content", "text")
):
made_progress = True
for tc in (getattr(delta, "tool_calls", None) or []):
idx = getattr(tc, "index", 0) or 0
acc = self.tool_calls_acc.setdefault(
@@ -9620,6 +10003,7 @@ class _ChatStreamAccumulator:
content="".join(self.content_parts),
tool_calls=tool_calls,
reasoning="".join(self.reasoning_parts) or None,
reasoning_details=self.reasoning_details or None,
)
choice = SimpleNamespace(
index=0,
@@ -10081,12 +10465,20 @@ def _call_llm_impl(
# fall straight through to provider/model fallback; fast blips (a
# streaming-close or a 5xx) still retry, since those are cheap.
if task == "compression" and _is_timeout_error(transient_err):
logger.info(
"Auxiliary compression: timeout on the critical path; "
"skipping same-provider retry and falling back: %s",
transient_err,
)
raise
# A fast first-token fail (dead stream detected within the
# 60s no-progress window, zero output seen) is cheap — take
# the normal same-provider retry chain first; the provider
# is often fine and only that one stream was stillborn. A
# mid-stream stall or hard-ceiling timeout skips straight to
# fallback, because re-running a multi-minute summary on the
# same provider doubles the user-visible stall (#54465).
if "no-progress timeout" not in str(transient_err):
logger.info(
"Auxiliary compression: timeout on the critical path; "
"skipping same-provider retry and falling back: %s",
transient_err,
)
raise
_max_transient_retries = _transient_retry_count()
_last_transient = transient_err
for _attempt in range(1, _max_transient_retries + 1):
@@ -10582,7 +10974,38 @@ def _call_llm_impl(
raise
def extract_content_or_reasoning(response) -> str:
def _coerce_llm_message(response):
"""Pull a message (dict, object, or str) out of a response-or-message value.
Compression and some OpenAI-compatible proxies hand us a dict-shaped
response or a bare message; vision/oneshot callers pass a ChatCompletion
object. MagicMock ``reasoning_*`` attrs are not strings — callers that
want the empty-content failure path rely on that.
"""
if response is None or isinstance(response, str):
return response
if isinstance(response, dict):
if "choices" not in response:
return response
choices = response.get("choices") or []
if not choices:
return None
first = choices[0]
return first.get("message") if isinstance(first, dict) else getattr(first, "message", None)
choices = getattr(response, "choices", None)
if not choices:
return response
first = choices[0]
return first.get("message") if isinstance(first, dict) else getattr(first, "message", None)
def _message_field(msg, name):
if isinstance(msg, dict):
return msg.get(name)
return getattr(msg, name, None)
def extract_content_or_reasoning(response, *, max_reasoning_chars: int | None = None) -> str:
"""Extract content from an LLM response, falling back to reasoning fields.
Mirrors the main agent loop's behavior when a reasoning model (DeepSeek-R1,
@@ -10595,12 +11018,24 @@ def extract_content_or_reasoning(response) -> str:
structured reasoning fields (DeepSeek, Moonshot, NovitaAI, etc.).
3. ``message.reasoning_details`` — OpenRouter unified array format.
Accepts a full response or a bare message (dict or object). When
``max_reasoning_chars`` is set, a reasoning-field fallback is truncated
so an unbounded chain-of-thought cannot become the compaction summary.
Returns the best available text, or ``""`` if nothing found.
"""
import re
msg = response.choices[0].message
content = (msg.content or "").strip()
msg = _coerce_llm_message(response)
if msg is None:
return ""
if isinstance(msg, str):
return msg.strip()
raw = _message_field(msg, "content")
if not isinstance(raw, str):
raw = str(raw) if raw else ""
content = raw.strip()
if content:
# Strip inline think/reasoning blocks (mirrors _strip_think_blocks)
@@ -10616,11 +11051,11 @@ def extract_content_or_reasoning(response) -> str:
# Content is empty or reasoning-only — try structured reasoning fields
reasoning_parts: list[str] = []
for field in ("reasoning", "reasoning_content"):
val = getattr(msg, field, None)
val = _message_field(msg, field)
if val and isinstance(val, str) and val.strip() and val not in reasoning_parts:
reasoning_parts.append(val.strip())
details = getattr(msg, "reasoning_details", None)
details = _message_field(msg, "reasoning_details")
if details and isinstance(details, list):
for detail in details:
if isinstance(detail, dict):
@@ -10632,10 +11067,18 @@ def extract_content_or_reasoning(response) -> str:
if summary and summary not in reasoning_parts:
reasoning_parts.append(summary.strip() if isinstance(summary, str) else str(summary))
if reasoning_parts:
return "\n\n".join(reasoning_parts)
if not reasoning_parts:
return ""
return ""
text = "\n\n".join(reasoning_parts)
if max_reasoning_chars is not None and len(text) > max_reasoning_chars:
logger.warning(
"fell back to reasoning fields (%d chars); truncating to %d",
len(text),
max_reasoning_chars,
)
return text[:max_reasoning_chars]
return text
@_relay_auxiliary_call_async
+12 -19
View File
@@ -1660,16 +1660,14 @@ def _run_review_in_thread(
# summary still needs the completed review agent's tool results.
review_messages = list(getattr(review_agent, "_session_messages", []))
# Tear down memory providers while stdout is still
# redirected so background thread teardown (Honcho flush,
# Hindsight sync, etc.) stays silent. The finally block
# below is a safety net for the exception path.
# The fork shares the foreground session ID for prompt-cache
# parity. Do not call close() or shutdown_memory_provider():
# both are session-bound lifecycle operations, and close() also
# kills registered terminal processes and cleans environments for
# that ID. Releasing only this fork's clients leaves the live
# session and its child processes untouched.
try:
review_agent.shutdown_memory_provider()
except Exception:
pass
try:
review_agent.close()
review_agent.release_clients()
except Exception:
pass
review_agent = None
@@ -1728,11 +1726,10 @@ def _run_review_in_thread(
_log_review_completion(review_usage, "error")
agent._emit_auxiliary_failure("background review", e)
finally:
# Safety-net cleanup for the exception path. Normal completion already
# shut down inside the thread-scoped silence above. Re-enter the
# thread-scoped silence here so teardown output (Honcho flush, Hindsight
# sync, background thread joins) stays quiet even on the exception path,
# without blanking other threads' streams.
# Safety-net cleanup for the exception path. Normal completion already
# released its clients inside the thread-scoped silence above. Re-enter
# the thread-scoped silence here so exception-path cleanup output stays
# quiet without blanking other threads' streams.
# Also a safety-net completion: covers exceptions raised during setup
# before the request-phase finally. Both tracking cleanup and the
# per-run completion publication are identity-scoped and idempotent.
@@ -1741,11 +1738,7 @@ def _run_review_in_thread(
try:
with thread_scoped_silence():
try:
review_agent.shutdown_memory_provider()
except Exception:
pass
try:
review_agent.close()
review_agent.release_clients()
except Exception:
pass
except Exception:
+143 -22
View File
@@ -27,6 +27,7 @@ the same Converse API integration in TypeScript via ``@aws-sdk/client-bedrock``.
Requires: ``boto3`` (optional dependency — only needed when using the Bedrock provider).
"""
import base64
import json
import logging
import os
@@ -1007,28 +1008,84 @@ def convert_messages_to_converse(
if role == "assistant":
content_blocks = []
# Convert text content
if isinstance(content, str) and content.strip():
content_blocks.append({"text": content})
elif isinstance(content, list):
content_blocks.extend(_convert_content_to_converse(content))
ordered_blocks = msg.get("bedrock_content_blocks")
if isinstance(ordered_blocks, list) and ordered_blocks:
# Rebuild the exact Bedrock block sequence captured at
# normalization time. Redacted bytes are stored as base64 so
# the sidecar remains JSON-safe in assistant history.
for block in ordered_blocks:
if not isinstance(block, dict):
continue
if "text" in block and isinstance(block["text"], str):
content_blocks.append({"text": block["text"]})
elif "reasoningContent" in block:
reasoning = block["reasoningContent"]
if not isinstance(reasoning, dict):
continue
replay = {}
if isinstance(reasoning.get("text"), str):
replay["text"] = reasoning["text"]
encoded = reasoning.get("redactedContentBase64")
if isinstance(encoded, str) and encoded:
try:
replay["redactedContent"] = base64.b64decode(encoded, validate=True)
except (ValueError, TypeError):
continue
if replay:
content_blocks.append({"reasoningContent": replay})
elif "toolUse" in block and isinstance(block["toolUse"], dict):
tu = block["toolUse"]
content_blocks.append({"toolUse": {
"toolUseId": tu.get("toolUseId", ""),
"name": tu.get("name", ""),
"input": tu.get("input", {}),
}})
# Convert tool calls
tool_calls = msg.get("tool_calls", [])
for tc in (tool_calls or []):
fn = tc.get("function", {})
args_str = fn.get("arguments", "{}")
try:
args_dict = json.loads(args_str) if isinstance(args_str, str) else args_str
except (json.JSONDecodeError, TypeError):
args_dict = {}
content_blocks.append({
"toolUse": {
"toolUseId": tc.get("id", ""),
"name": fn.get("name", ""),
"input": args_dict,
}
})
if not content_blocks:
ordered_blocks = None
if content_blocks:
# Ordered replay is authoritative; do not append parallel
# reasoning/text/tool lists a second time.
pass
else:
# Bedrock may return opaque encrypted reasoning instead of text.
# Preserve the payload in the provider-neutral reasoning_details
# envelope so the next tool turn can replay it byte-for-byte.
for detail in (msg.get("reasoning_details") or []):
if not isinstance(detail, dict) or detail.get("type") != "redacted_thinking":
continue
encoded = detail.get("data") or detail.get("redactedContentBase64")
if not isinstance(encoded, str) or not encoded:
continue
try:
redacted = base64.b64decode(encoded, validate=True)
except (ValueError, TypeError):
continue
content_blocks.append({"reasoningContent": {"redactedContent": redacted}})
# Convert text content
if isinstance(content, str) and content.strip():
content_blocks.append({"text": content})
elif isinstance(content, list):
content_blocks.extend(_convert_content_to_converse(content))
# Convert tool calls
tool_calls = msg.get("tool_calls", [])
for tc in (tool_calls or []):
fn = tc.get("function", {})
args_str = fn.get("arguments", "{}")
try:
args_dict = json.loads(args_str) if isinstance(args_str, str) else args_str
except (json.JSONDecodeError, TypeError):
args_dict = {}
content_blocks.append({
"toolUse": {
"toolUseId": tc.get("id", ""),
"name": fn.get("name", ""),
"input": args_dict,
}
})
if not content_blocks:
content_blocks = [{"text": _EMPTY_TEXT_PLACEHOLDER}]
@@ -1102,19 +1159,48 @@ def normalize_converse_response(response: Dict) -> SimpleNamespace:
text_parts = []
reasoning_parts = []
reasoning_details = []
ordered_blocks = []
tool_calls = []
for block in content_blocks:
if "text" in block:
text_parts.append(block["text"])
ordered_blocks.append({"text": block["text"]})
elif "reasoningContent" in block:
reasoning = block["reasoningContent"]
if isinstance(reasoning, dict):
thinking_text = reasoning.get("text", "")
encoded = None
if thinking_text:
reasoning_parts.append(str(thinking_text))
redacted = reasoning.get("redactedContent")
if redacted is not None:
if isinstance(redacted, (bytes, bytearray)):
encoded = base64.b64encode(bytes(redacted)).decode("ascii")
elif isinstance(redacted, str):
encoded = redacted
else:
encoded = None
if encoded:
reasoning_details.append({
"type": "redacted_thinking",
"data": encoded,
})
if thinking_text or encoded:
ordered_reasoning = {}
if thinking_text:
ordered_reasoning["text"] = str(thinking_text)
if encoded:
ordered_reasoning["redactedContentBase64"] = encoded
ordered_blocks.append({"reasoningContent": ordered_reasoning})
elif "toolUse" in block:
tu = block["toolUse"]
ordered_blocks.append({"toolUse": {
"toolUseId": tu.get("toolUseId", ""),
"name": tu.get("name", ""),
"input": tu.get("input", {}),
}})
tool_calls.append(SimpleNamespace(
id=tu.get("toolUseId", ""),
type="function",
@@ -1130,6 +1216,8 @@ def normalize_converse_response(response: Dict) -> SimpleNamespace:
content="\n".join(text_parts) if text_parts else None,
tool_calls=tool_calls if tool_calls else None,
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
reasoning_details=reasoning_details or None,
bedrock_content_blocks=ordered_blocks or None,
)
# Build usage stats. Converse's inputTokens excludes cache read/write
@@ -1226,7 +1314,10 @@ def stream_converse_with_callbacks(
"""
text_parts: List[str] = []
reasoning_parts: List[str] = []
reasoning_details: List[Dict[str, Any]] = []
tool_calls: List[SimpleNamespace] = []
stream_blocks: Dict[int, Dict[str, Any]] = {}
current_block_index: Optional[int] = None
current_tool: Optional[Dict] = None
current_text_buffer: List[str] = []
has_tool_use = False
@@ -1248,7 +1339,9 @@ def stream_converse_with_callbacks(
break
if "contentBlockStart" in event:
start = event["contentBlockStart"].get("start", {})
start_event = event["contentBlockStart"]
current_block_index = start_event.get("contentBlockIndex", len(stream_blocks))
start = start_event.get("start", {})
if "toolUse" in start:
has_tool_use = True
# Flush any accumulated text
@@ -1260,6 +1353,11 @@ def stream_converse_with_callbacks(
"name": start["toolUse"].get("name", ""),
"input_json": "",
}
stream_blocks[current_block_index] = {"toolUse": {
"toolUseId": current_tool["toolUseId"],
"name": current_tool["name"],
"input": {},
}}
if on_tool_start:
on_tool_start(current_tool["name"])
@@ -1267,6 +1365,8 @@ def stream_converse_with_callbacks(
delta = event["contentBlockDelta"].get("delta", {})
if "text" in delta:
text = delta["text"]
block = stream_blocks.setdefault(current_block_index if current_block_index is not None else len(stream_blocks), {"text": ""})
block["text"] = block.get("text", "") + text
current_text_buffer.append(text)
# Fire text delta callback only when no tool calls are present
# (same semantics as Anthropic/chat_completions streaming)
@@ -1284,6 +1384,23 @@ def stream_converse_with_callbacks(
reasoning_parts.append(str(thinking_text))
if on_reasoning_delta:
on_reasoning_delta(thinking_text)
block = stream_blocks.setdefault(current_block_index if current_block_index is not None else len(stream_blocks), {"reasoningContent": {}})
block.setdefault("reasoningContent", {})["text"] = block["reasoningContent"].get("text", "") + str(thinking_text)
redacted = reasoning.get("redactedContent")
if redacted is not None:
if isinstance(redacted, (bytes, bytearray)):
encoded = base64.b64encode(bytes(redacted)).decode("ascii")
elif isinstance(redacted, str):
encoded = redacted
else:
encoded = None
if encoded:
reasoning_details.append({
"type": "redacted_thinking",
"data": encoded,
})
block = stream_blocks.setdefault(current_block_index if current_block_index is not None else len(stream_blocks), {"reasoningContent": {}})
block.setdefault("reasoningContent", {})["redactedContentBase64"] = encoded
elif "contentBlockStop" in event:
if current_tool is not None:
@@ -1299,6 +1416,8 @@ def stream_converse_with_callbacks(
arguments=json.dumps(input_dict),
),
))
if current_block_index is not None and current_block_index in stream_blocks:
stream_blocks[current_block_index]["toolUse"]["input"] = input_dict
current_tool = None
elif current_text_buffer:
text_parts.append("".join(current_text_buffer))
@@ -1325,6 +1444,8 @@ def stream_converse_with_callbacks(
content="\n".join(text_parts) if text_parts else None,
tool_calls=tool_calls if tool_calls else None,
reasoning_content="\n\n".join(reasoning_parts) if reasoning_parts else None,
reasoning_details=reasoning_details or None,
bedrock_content_blocks=[stream_blocks[i] for i in sorted(stream_blocks)] or None,
)
input_tokens = usage_data.get("inputTokens", 0)
+406 -143
View File
@@ -1055,6 +1055,59 @@ def should_use_direct_api_call(agent) -> bool:
_DIRECT_API_ACTIVITY_HEARTBEAT_SECONDS = 15.0
def _managed_local_load_notice(agent, api_kwargs: dict) -> "Optional[str]":
"""A live phase notice while the managed local server works before the
first token, or None when neither phase (nor the managed server) applies:
- "⏳ loading <model> into memory — N%" (weights streaming off disk;
real per-tensor percent from the router's SSE stream)
- "⚙ processing prompt — N of ~M tokens (P%)" (prefill; live counter
from /slots, denominator estimated from the request body)
A cold local model spends ~tens of seconds loading and a long-context
turn spends tens more in prefill; without this, both windows render as
the generic "no output yet (provider may be slow or overloaded)" stall
warning — alarming copy for healthy, expected phases.
"""
try:
base = str(getattr(agent, "base_url", "") or "")
if not base:
return None
import json as _json
from urllib.parse import urlparse
from hermes_cli.local_runtime.load_progress import (
get_loading_progress,
get_prefill_progress,
)
from hermes_cli.local_runtime.supervisor import state_path
state = _json.loads(state_path().read_text(encoding="utf-8"))
managed = urlparse(str(state.get("base_url", ""))).netloc.lower()
if not managed or urlparse(base).netloc.lower() != managed:
return None
model = str(api_kwargs.get("model", ""))
progress = get_loading_progress().get(model)
if progress is not None:
return (
f"⏳ loading {model} into memory — {progress['percent']}% "
"(responses start once the model is loaded)"
)
prefill = get_prefill_progress(model)
if prefill is not None:
processed = int(prefill["processed"])
total = estimate_request_context_tokens(api_kwargs)
if total and total >= processed:
pct = max(0, min(100, round(processed / total * 100)))
return f"⚙ processing prompt — {pct}%"
# Counter past the estimate (estimator undercounted): no honest
# denominator, so no percent — the UI shows label-only.
return "⚙ processing prompt"
return None
except Exception: # noqa: BLE001 — a status nicety must never break a call
return None
def _resolve_direct_stale_timeout(agent, api_kwargs: dict) -> float:
"""Stale budget for the inline non-streaming call.
@@ -1957,6 +2010,54 @@ def build_api_kwargs(agent, api_messages: list, tools_for_api: list | None = Non
# ── chat_completions (default) ─────────────────────────────────────
_ct = agent._get_transport()
# xAI's chat-completions endpoint reserves the function name
# ``tool_search`` for its native server-side tool and rejects the whole
# request when the client Tool Search bridge declares it (HTTP 400
# "The function name tool_search is reserved for the tool_search tool",
# #95003) — same reserved-name class the codex_responses branch above
# already sanitizes tools for (#27197). Rename the bridge's wire
# declaration to an alias; normalize_response maps model calls back.
# Deep-copy first (the #27907 in-place-mutation lesson): tools_for_api
# aliases agent.tools, and renaming in place would corrupt the shared
# per-agent tool registry for every later non-xAI request.
_is_xai_chat = (
agent.provider in {"xai", "xai-oauth"}
or agent._base_url_hostname == "api.x.ai"
)
# Reset request-local alias provenance for THIS request; the rewrite
# below repopulates it when it actually emits aliases. Without the
# reset, a stale map from an earlier request on the same transport
# could reverse-map a name this request never aliased.
if _ct is not None and hasattr(_ct, "_last_wire_aliases"):
_ct._last_wire_aliases = {}
if _is_xai_chat and tools_for_api:
try:
import copy as _copy_xai
from agent.transports.chat_completions import (
_rename_tool_search_bridge_for_xai,
)
_has_bridge = any(
(t.get("function") or {}).get("name") == "tool_search"
for t in tools_for_api
if isinstance(t, dict)
)
if _has_bridge:
tools_for_api = _copy_xai.deepcopy(tools_for_api)
tools_for_api, _xai_alias_map = _rename_tool_search_bridge_for_xai(
tools_for_api
)
# Record provenance so normalize_response reverses ONLY the
# aliases this request put on the wire.
if _ct is not None:
_ct._last_wire_aliases = _xai_alias_map
except Exception as exc:
logger.warning(
"%s⚠️ Failed to alias tool_search bridge for xAI: %s",
getattr(agent, "log_prefix", ""), exc,
)
# Provider detection flags
_is_qwen = agent._is_qwen_portal()
_is_or = agent._is_openrouter_url()
@@ -2277,6 +2378,10 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic
if ordered_blocks:
msg["anthropic_content_blocks"] = ordered_blocks
bedrock_blocks = getattr(assistant_message, "bedrock_content_blocks", None)
if bedrock_blocks:
msg["bedrock_content_blocks"] = bedrock_blocks
# Codex Responses API: preserve encrypted reasoning items for
# multi-turn continuity. These get replayed as input on the next turn.
codex_items = getattr(assistant_message, "codex_reasoning_items", None)
@@ -3017,7 +3122,7 @@ def handle_max_iterations(agent, messages: list, api_call_count: int) -> str:
# timestamp (preserved on gateway user replay entries for the
# stale-confirmation expiry check — #47868 rejection class),
# and every Hermes-internal underscore-prefixed scaffolding key.
for schema_foreign in ("tool_name", "codex_reasoning_items", "codex_message_items", "timestamp"):
for schema_foreign in ("tool_name", "codex_reasoning_items", "codex_message_items", "timestamp", "platform_message_id"):
api_msg.pop(schema_foreign, None)
# api_content (the persist-what-you-send sidecar) carries the
# exact bytes every main-loop call sent for this message —
@@ -3441,14 +3546,11 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
if emit is not None:
emit(final_text=final_text, finished=finished, error=error)
# Cron and other non-interactive, nested-pool contexts deadlock on the
# spawned worker thread (#62151). They also have no stream consumer, so the
# deltas this path produces go nowhere. Delegate to the non-streaming entry
# (which runs inline via should_use_direct_api_call) exactly like the codex
# branch below — routing through the _interruptible_api_call method keeps the
# outer loop's per-request retry/refresh seam intact.
if should_use_direct_api_call(agent):
return agent._interruptible_api_call(api_kwargs)
# Cron turns and delegated children (should_use_direct_api_call) used to be
# short-circuited here onto the NON-streaming wire. They now stay on this
# streaming path and run the request inline — see the ``_inline`` block
# before the poll loop below. Only the codex branch still detours through
# _interruptible_api_call (it streams internally).
if agent.api_mode == "codex_responses":
# Codex streams internally via _run_codex_stream. The main dispatch
@@ -4078,6 +4180,29 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
provider_tool_in_flight["yes"] = True
except Exception:
pass
# Payload-empty terminal chunk: the provider completed the
# stream (`finish_reason` set, no further writable delta). The
# attempt/writer fence exists to stop a superseded stream from
# writing *more* text. Fending this marker-only chunk discards
# the only completion signal, which the drop-guard then
# mislabels as a mid-stream drop. A finish chunk that still
# carries content/tool_calls remains gated.
try:
_choices = getattr(_chunk, "choices", None)
if _choices:
_choice = _choices[0]
if getattr(_choice, "finish_reason", None):
_delta = getattr(_choice, "delta", None)
_has_write = bool(
getattr(_delta, "content", None)
or getattr(_delta, "tool_calls", None)
or getattr(_delta, "reasoning_content", None)
or getattr(_delta, "reasoning", None)
)
if not _has_write:
return True
except Exception:
pass
if not _stream_attempt_is_active(stream_attempt_id):
return False
token = _writer_token["value"]
@@ -4249,12 +4374,36 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
),
raw_text=f"{_err_type}: {_err_msg}",
)
# Nous Portal usage frames often have choices=[] plus
# lastOne=true and no [DONE]. Treat that as a clean
# terminal, not a mid-stream drop (#90848).
last_one = getattr(chunk, "lastOne", None)
if last_one is None:
extra = getattr(chunk, "model_extra", None)
if isinstance(extra, dict):
last_one = extra.get("lastOne")
# Integer/string-truthy sentinels included — relabelled
# upstreams have been seen sending 1 / "true".
if last_one in (True, 1, "true") and finish_reason is None:
finish_reason = "stop"
continue
delta = chunk.choices[0].delta
if hasattr(chunk, "model") and chunk.model:
model_name = chunk.model
# Extract terminal chunk fields BEFORE any content-shape `continue`.
# Backends that merge finish_reason into the final content chunk
# (e.g. vLLM >= 0.1.dev20051) can have that chunk swallowed by the
# SSE-echo guard below when the tokenizer emits standalone ':'
# tokens — the finish chunk would never register and the response
# would be falsely flagged as truncated (#94614).
chunk_finish_reason = getattr(chunk.choices[0], "finish_reason", None)
if chunk_finish_reason:
finish_reason = chunk_finish_reason
if hasattr(chunk, "usage") and chunk.usage:
usage_obj = chunk.usage
# Accumulate reasoning content
reasoning_text = getattr(delta, "reasoning_content", None) or getattr(delta, "reasoning", None)
if reasoning_text:
@@ -4390,13 +4539,10 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
# discarding the attempted action.
result["partial_tool_names"].append(name)
chunk_finish_reason = getattr(chunk.choices[0], "finish_reason", None)
if chunk_finish_reason:
finish_reason = chunk_finish_reason
# Usage in the final chunk
if hasattr(chunk, "usage") and chunk.usage:
usage_obj = chunk.usage
# (finish_reason/usage are now extracted at the top of the loop
# body. The old tail-side extraction sat after the SSE-echo
# guard's `continue` paths, so a merged finish chunk that tripped
# the guard never registered — false "stream truncated".)
_close_managed_stream()
@@ -4547,10 +4693,16 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
# text content but no tool calls. Without this guard the partial
# text is silently stamped finish_reason="stop" and the turn ends as
# if complete — the model's intended next step is lost (#32086).
# When `include_usage` is requested, OpenAI-compliant providers (e.g.
# vLLM, OpenAI, DeepSeek) emit a final usage-only chunk with empty
# choices and no finish_reason. Receiving a valid usage object proves
# the provider completed generation and closed the stream cleanly
# (#91373), so it is not a mid-stream drop.
_text_only_dropped_no_finish = (
finish_reason is None
and content_parts
and not tool_calls_acc
and usage_obj is None
)
if _text_only_dropped_no_finish:
logger.warning(
@@ -5216,143 +5368,246 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
if _reasoning_floor is not None:
_stream_stale_timeout = max(_stream_stale_timeout, _reasoning_floor)
t = threading.Thread(target=_context_thread_target(_call), daemon=True)
t.start()
# Delegated children and gateway cron turns run the streaming request
# INLINE on the conversation thread: spawning the interrupt worker inside
# their nested thread pools wedges before the socket opens (#62151,
# #60203). They used to be routed to the non-streaming wire for that
# reason — but streaming is also the transport keepalive and the
# liveness signal: a non-streaming POST that stays silent through a
# reasoning model's thinking phase is killed by edge proxies (z.ai 524,
# #90202) and by our own stale watchdog, which cannot tell thinking from
# a hang when no bytes ever arrive (#100260). Inline mode keeps the
# stream (per-token liveness) and moves ONLY the lightweight poll loop
# below — heartbeat, stale detector, interrupt abort — onto a monitor
# thread. The monitor never issues a request, so the no-worker property
# that fixes the deadlock class is preserved (same shape as the
# direct_api_call watchdog timer).
_inline = should_use_direct_api_call(agent)
_call_done = threading.Event()
_monitor_interrupted = {"yes": False}
def _run_call():
try:
_call()
finally:
_call_done.set()
if _inline:
t = None
else:
t = threading.Thread(target=_context_thread_target(_run_call), daemon=True)
t.start()
def _call_alive() -> bool:
return not _call_done.is_set()
def _wait_call(timeout: float) -> None:
_call_done.wait(timeout=timeout)
_last_heartbeat = time.time()
_HEARTBEAT_INTERVAL = 30.0 # seconds between gateway activity touches
while t.is_alive():
t.join(timeout=0.3)
# Managed local server: a cold model streams weights off disk for tens
# of seconds before the first token can exist. Surface THAT immediately
# (real per-tensor percent from the router's SSE stream) instead of
# letting the wait fall through to the 30s "provider may be slow or
# overloaded" copy. Checked on a ~1s cadence only while no chunks have
# arrived; the probe is an in-memory snapshot read, not a network call.
_last_load_poll = 0.0
_load_notice_shown = False
_load_notice_misses = 0
_is_local_base = bool(agent.base_url) and is_local_endpoint(agent.base_url)
# Periodic heartbeat: touch the agent's activity tracker so the
# gateway's inactivity monitor knows we're alive while waiting
# for stream chunks. Without this, long thinking pauses (e.g.
# reasoning models) or slow prefill on local providers (Ollama)
# trigger false inactivity timeouts. The _call thread touches
# activity on each chunk, but the gap between API call start
# and first chunk can exceed the gateway timeout — especially
# when the stale-stream timeout is disabled (local providers).
_hb_now = time.time()
if _hb_now - _last_heartbeat >= _HEARTBEAT_INTERVAL:
_last_heartbeat = _hb_now
_waiting_secs = int(_hb_now - last_chunk_time["t"])
if _waiting_secs >= _HEARTBEAT_INTERVAL:
# No chunks for 30s+ — rewrite the live spinner/status line
# so CLI/TUI/Desktop users see WHAT the wait is (slow or
# overloaded provider / long thinking pause) instead of an
# unexplained generic spinner, and WHEN recovery kicks in.
if (
_stream_stale_timeout is not None
and _stream_stale_timeout != float("inf")
):
_recovery = f"; auto-reconnect at {int(_stream_stale_timeout)}s"
def _monitor_loop() -> None:
nonlocal _last_heartbeat, _last_load_poll, _load_notice_shown, _load_notice_misses
while _call_alive():
_wait_call(0.3)
_hb_now = time.time()
# Cold-load window: last_chunk_time is touched at request-client
# creation and then only by REAL chunks, so "no chunk for 2s+" is
# true through a model load (nothing can stream while the child is
# still mapping weights) and false during healthy token flow —
# which is what keeps this poll off the streaming hot path. The
# probe itself is an in-memory snapshot read.
if (
_is_local_base
and _hb_now - last_chunk_time["t"] >= 2.0
and _hb_now - _last_load_poll >= 1.0
):
_last_load_poll = _hb_now
_load_notice = _managed_local_load_notice(agent, api_kwargs)
if _load_notice is not None:
agent._emit_wait_notice(_load_notice)
agent._touch_activity("local model loading")
_load_notice_shown = True
_load_notice_misses = 0
# Loading IS liveness for the heartbeat; the stale detector
# needs no help — the local floor (900s) dwarfs any load.
_last_heartbeat = _hb_now
continue
if _load_notice_shown:
# One missed sample is routine (a /slots read straddling a
# batch boundary, a 2s probe timeout under load) — clearing
# on it made the status line strobe blank once every few
# seconds mid-prefill. Only a SUSTAINED absence means the
# phase really ended.
_load_notice_misses += 1
if _load_notice_misses >= 3:
_load_notice_shown = False
_load_notice_misses = 0
agent._emit_wait_notice("")
# Periodic heartbeat: touch the agent's activity tracker so the
# gateway's inactivity monitor knows we're alive while waiting
# for stream chunks. Without this, long thinking pauses (e.g.
# reasoning models) or slow prefill on local providers (Ollama)
# trigger false inactivity timeouts. The _call thread touches
# activity on each chunk, but the gap between API call start
# and first chunk can exceed the gateway timeout — especially
# when the stale-stream timeout is disabled (local providers).
if _hb_now - _last_heartbeat >= _HEARTBEAT_INTERVAL:
_last_heartbeat = _hb_now
_waiting_secs = int(_hb_now - last_chunk_time["t"])
if _waiting_secs >= _HEARTBEAT_INTERVAL:
# No chunks for 30s+ — rewrite the live spinner/status line
# so CLI/TUI/Desktop users see WHAT the wait is (slow or
# overloaded provider / long thinking pause) instead of an
# unexplained generic spinner, and WHEN recovery kicks in.
if (
_stream_stale_timeout is not None
and _stream_stale_timeout != float("inf")
):
_recovery = f"; auto-reconnect at {int(_stream_stale_timeout)}s"
else:
_recovery = ""
agent._emit_wait_notice(
f"⏳ waiting on {api_kwargs.get('model', 'the provider')} — "
f"{_waiting_secs}s with no output yet (provider may be "
f"slow or overloaded, or the model is thinking{_recovery})"
)
else:
_recovery = ""
# Chunks are flowing — keep the activity tracker fresh but
# leave the live display alone.
agent._touch_activity(
f"waiting for stream response ({_waiting_secs}s, no chunks yet)"
)
# Detect stale streams: connections kept alive by SSE pings
# but delivering no real chunks. Kill the client so the
# inner retry loop can start a fresh connection.
_stale_elapsed = time.time() - last_chunk_time["t"]
if _stale_elapsed > _stream_stale_timeout:
_est_ctx = estimate_request_context_tokens(api_kwargs)
logger.warning(
"Stream stale for %.0fs (threshold %.0fs) — no chunks received. "
"model=%s context=~%s tokens. Killing connection.",
_stale_elapsed, _stream_stale_timeout,
api_kwargs.get("model", "unknown"), f"{_est_ctx:,}",
)
agent._buffer_status(
f"⚠️ No response from provider for {int(_stale_elapsed)}s "
f"(model: {api_kwargs.get('model', 'unknown')}, "
f"context: ~{_est_ctx:,} tokens). "
f"Reconnecting..."
)
try:
_cancel_current_stream_attempt("stale_stream_kill")
_close_request_client_once("stale_stream_kill")
except Exception:
pass
# Circuit breaker (#58962): count the stale kill. See the
# canonical comment block above ``_stale_streak()``.
_bump_stale_streak(agent)
# Rebuild the primary client too — its connection pool
# may hold dead sockets from the same provider outage.
if agent.api_mode == "anthropic_messages":
# #67142: the stale stream ran on a request-local anthropic
# client, already socket-aborted above via
# _close_request_client_once (which unblocks the worker and
# preserves the #28161 no-hang guarantee). The shared
# _anthropic_client is NOT the in-flight transport, so we must
# not close it from this poll (stranger) thread — that was the
# FD-recycle corruption vector. Nothing further is needed.
pass
else:
# #70773: same FD-recycle corruption vector as #67142.
# The shared OpenAI client's connection pool must NOT be
# closed from this watchdog/poll thread — worker threads
# from previous stale-killed attempts may still be
# unwinding their SSL BIOs. The request-local client is
# already closed above via _close_request_client_once.
# The shared client will be replaced lazily by
# _ensure_primary_openai_client on the next request.
pass
# Reset the timer so we don't kill repeatedly while
# the inner thread processes the closure.
last_chunk_time["t"] = time.time()
agent._emit_wait_notice(
f"⏳ waiting on {api_kwargs.get('model', 'the provider')} — "
f"{_waiting_secs}s with no output yet (provider may be "
f"slow or overloaded, or the model is thinking{_recovery})"
f"⚠ no output from provider for {int(_stale_elapsed)}s — "
f"reconnecting..."
)
else:
# Chunks are flowing — keep the activity tracker fresh but
# leave the live display alone.
agent._touch_activity(
f"waiting for stream response ({_waiting_secs}s, no chunks yet)"
f"stale stream detected after {int(_stale_elapsed)}s, reconnecting"
)
# Detect stale streams: connections kept alive by SSE pings
# but delivering no real chunks. Kill the client so the
# inner retry loop can start a fresh connection.
_stale_elapsed = time.time() - last_chunk_time["t"]
if _stale_elapsed > _stream_stale_timeout:
_est_ctx = estimate_request_context_tokens(api_kwargs)
logger.warning(
"Stream stale for %.0fs (threshold %.0fs) — no chunks received. "
"model=%s context=~%s tokens. Killing connection.",
_stale_elapsed, _stream_stale_timeout,
api_kwargs.get("model", "unknown"), f"{_est_ctx:,}",
)
agent._buffer_status(
f"⚠️ No response from provider for {int(_stale_elapsed)}s "
f"(model: {api_kwargs.get('model', 'unknown')}, "
f"context: ~{_est_ctx:,} tokens). "
f"Reconnecting..."
)
try:
_cancel_current_stream_attempt("stale_stream_kill")
_close_request_client_once("stale_stream_kill")
except Exception:
pass
# Circuit breaker (#58962): count the stale kill. See the
# canonical comment block above ``_stale_streak()``.
_bump_stale_streak(agent)
# Rebuild the primary client too — its connection pool
# may hold dead sockets from the same provider outage.
if agent.api_mode == "anthropic_messages":
# #67142: the stale stream ran on a request-local anthropic
# client, already socket-aborted above via
# _close_request_client_once (which unblocks the worker and
# preserves the #28161 no-hang guarantee). The shared
# _anthropic_client is NOT the in-flight transport, so we must
# not close it from this poll (stranger) thread — that was the
# FD-recycle corruption vector. Nothing further is needed.
pass
else:
# #70773: same FD-recycle corruption vector as #67142.
# The shared OpenAI client's connection pool must NOT be
# closed from this watchdog/poll thread — worker threads
# from previous stale-killed attempts may still be
# unwinding their SSL BIOs. The request-local client is
# already closed above via _close_request_client_once.
# The shared client will be replaced lazily by
# _ensure_primary_openai_client on the next request.
pass
# Reset the timer so we don't kill repeatedly while
# the inner thread processes the closure.
last_chunk_time["t"] = time.time()
agent._emit_wait_notice(
f"⚠ no output from provider for {int(_stale_elapsed)}s — "
f"reconnecting..."
)
agent._touch_activity(
f"stale stream detected after {int(_stale_elapsed)}s, reconnecting"
)
if agent._interrupt_requested:
# The stale branch above already counted this iteration when its
# deadline won the race; do not double-count a simultaneous stop.
if _stale_elapsed <= _stream_stale_timeout:
_record_interrupted_provider_wait(
agent,
_stale_elapsed,
response_started=deltas_were_sent["yes"],
if agent._interrupt_requested:
# The stale branch above already counted this iteration when its
# deadline won the race; do not double-count a simultaneous stop.
if _stale_elapsed <= _stream_stale_timeout:
_record_interrupted_provider_wait(
agent,
_stale_elapsed,
response_started=deltas_were_sent["yes"],
)
# Mark THIS request cancelled before force-closing so the worker's
# exception handler recognizes the forced transport error as a
# cancel and exits without retrying or surfacing a network error.
# (#6600)
_request_cancelled["value"] = True
logger.debug(
"Force-closing streaming httpx client due to interrupt "
"(not a network error)."
)
# Mark THIS request cancelled before force-closing so the worker's
# exception handler recognizes the forced transport error as a
# cancel and exits without retrying or surfacing a network error.
# (#6600)
_request_cancelled["value"] = True
logger.debug(
"Force-closing streaming httpx client due to interrupt "
"(not a network error)."
)
try:
_cancel_current_stream_attempt("stream_interrupt_abort")
# #67142: kind-aware — anthropic aborts the request-local
# client's socket from this poll thread; the shared
# _anthropic_client is never closed here.
_close_request_client_once("stream_interrupt_abort")
except Exception:
pass
# Wait for the worker to unwind Relay-managed stream scopes
# (physical LLM + deferred logical) before surfacing
# InterruptedError. Raising immediately lets turn teardown
# (finish_logical_calls / end_turn / close_session) race a
# still-open physical scope and corrupt the LIFO stack —
# "scope handle is not at the top of the stack" → CLI EIO /
# redraw storm (#81521). No-op when Relay managed execution
# is not live.
_join_worker_for_relay_teardown(t, label="Streaming")
raise InterruptedError("Agent interrupted during streaming API call")
try:
_cancel_current_stream_attempt("stream_interrupt_abort")
# #67142: kind-aware — anthropic aborts the request-local
# client's socket from this poll thread; the shared
# _anthropic_client is never closed here.
_close_request_client_once("stream_interrupt_abort")
except Exception:
pass
# Wait for the worker to unwind Relay-managed stream scopes
# (physical LLM + deferred logical) before surfacing
# InterruptedError. Raising immediately lets turn teardown
# (finish_logical_calls / end_turn / close_session) race a
# still-open physical scope and corrupt the LIFO stack —
# "scope handle is not at the top of the stack" → CLI EIO /
# redraw storm (#81521). No-op when Relay managed execution
# is not live. (Inline mode has no worker: the request runs
# on the caller's thread and has already unwound by the time
# the InterruptedError below is raised.)
if t is not None:
_join_worker_for_relay_teardown(t, label="Streaming")
_monitor_interrupted["yes"] = True
return
if _inline:
# Request on THIS thread; heartbeat / stale / interrupt monitor on a
# side thread that only ever aborts sockets (never dispatches).
monitor = threading.Thread(
target=_context_thread_target(_monitor_loop),
name="stream-inline-monitor",
daemon=True,
)
monitor.start()
try:
_run_call()
finally:
monitor.join(timeout=2.0)
else:
_monitor_loop()
if _monitor_interrupted["yes"]:
raise InterruptedError("Agent interrupted during streaming API call")
# Worker thread exited before the main thread's poll loop could check
# the interrupt flag. If the worker returned early due to an interrupt
# (e.g. _call_anthropic() detected _interrupt_requested and returned
@@ -5464,6 +5719,14 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta=
# responsive. See the canonical comment block above ``_stale_streak()``.
if result["response"] is not None:
_reset_stale_streak(agent)
# Surface first-chunk timing for observability (forwarded to the
# ``post_api_request`` plugin hook by the conversation loop). The
# per-attempt stream diagnostic dict already records ``first_chunk_at``
# on the first received chunk; propagate the latest value onto the agent
# so the loop can read it without threading the diag through returns.
_diag_last = request_client_holder.get("diag")
if isinstance(_diag_last, dict) and _diag_last.get("first_chunk_at"):
agent._last_api_first_chunk_at = float(_diag_last["first_chunk_at"])
return result["response"]
# ── Provider fallback ──────────────────────────────────────────────────
+92
View File
@@ -0,0 +1,92 @@
"""Codex request identity helpers shared by agent client builders.
This leaf module intentionally has no dependency on the large auxiliary-client
router. Long-lived processes can therefore import a newly added client builder
without resolving a new symbol from an older cached ``auxiliary_client`` module.
"""
from __future__ import annotations
import base64
import json
from typing import Any, Dict
from urllib.parse import urlparse
CODEX_AUX_BASE_URL = "https://chatgpt.com/backend-api/codex"
def is_official_codex_base_url(base_url: str) -> bool:
"""Identify OpenAI's Codex endpoint without matching custom proxies."""
try:
parsed = urlparse(base_url)
path = parsed.path.rstrip("/")
return (
parsed.scheme == "https"
and parsed.hostname == "chatgpt.com"
and parsed.port in (None, 443)
and (path == "/backend-api/codex" or path.startswith("/backend-api/codex/"))
)
except (TypeError, ValueError):
return False
def codex_cloudflare_headers(
access_token: str, *, base_url: str = CODEX_AUX_BASE_URL,
) -> Dict[str, str]:
"""Identity and account headers for chatgpt.com/backend-api/codex.
OpenAI requires third-party harnesses to identify themselves. Requests to
the official endpoint always send Hermes' originator and version. Custom
endpoints retain the existing compatibility identity. In either case,
preserve ``ChatGPT-Account-ID`` from the OAuth JWT's
``chatgpt_account_id`` claim.
Malformed tokens are tolerated — we drop the account-ID header rather than
raise, so a bad token still surfaces as an auth error (401) instead of a
crash at client construction.
"""
headers = {
"User-Agent": "codex_cli_rs/0.0.0 (Hermes Agent)",
"originator": "codex_cli_rs",
}
if is_official_codex_base_url(base_url):
from hermes_cli import __version__
headers.update({
"User-Agent": f"HermesAgent/{__version__}",
"originator": "hermes-agent",
})
if not isinstance(access_token, str) or not access_token.strip():
return headers
try:
parts = access_token.split(".")
if len(parts) < 2:
return headers
payload_b64 = parts[1] + "=" * (-len(parts[1]) % 4)
claims = json.loads(base64.urlsafe_b64decode(payload_b64))
acct_id = claims.get("https://api.openai.com/auth", {}).get("chatgpt_account_id")
if isinstance(acct_id, str) and acct_id:
headers["ChatGPT-Account-ID"] = acct_id
except Exception:
pass
return headers
def apply_required_codex_headers(
client_kwargs: Dict[str, Any], *, access_token: str, base_url: str,
) -> None:
"""Keep required Codex identity after user/provider header overrides."""
if not is_official_codex_base_url(base_url):
return
required = codex_cloudflare_headers(access_token, base_url=base_url)
required_names = {name.lower() for name in required}
existing = client_kwargs.get("default_headers") or {}
client_kwargs["default_headers"] = {
**{
name: value
for name, value in existing.items()
if str(name).lower() not in required_names
},
**required,
}
+1
View File
@@ -310,6 +310,7 @@ def _record_codex_app_server_compaction(
# Native compaction rewrote the provider-side context; the usage anchor's
# transcript snapshot no longer matches what will be sent. Invalidate it.
agent._usage_anchor = None
agent._turn_base_usage_anchor = None
agent._last_compaction_in_place = False
try:
+12 -1
View File
@@ -135,9 +135,20 @@ def compute_session_context_breakdown(
# after the response) and far more accurate than the heuristic total.
from agent.model_metadata import anchored_context_tokens
# Prefer the turn-base anchor (first response of the current turn): on
# reasoning models, later same-turn responses inflate prompt_tokens with
# replayed thinking that evaporates at the turn boundary, so anchoring on
# the LAST response makes the meter sawtooth. Fall back to the last-
# response anchor, then to measured/estimated figures.
anchored_used = anchored_context_tokens(
messages or [], getattr(agent, "_usage_anchor", None)
messages or [],
getattr(agent, "_turn_base_usage_anchor", None),
charge_stale_thinking=False,
)
if anchored_used is None:
anchored_used = anchored_context_tokens(
messages or [], getattr(agent, "_usage_anchor", None)
)
measured_used = int(getattr(comp, "last_prompt_tokens", 0) or 0) if comp else 0
if anchored_used is not None:
context_used = anchored_used
+247 -44
View File
@@ -33,6 +33,7 @@ from agent.auxiliary_client import (
_is_connection_error,
aux_interrupt_protection,
call_llm,
extract_content_or_reasoning,
)
from agent.context_engine import ContextEngine, sanitize_memory_context
from agent.error_classifier import FailoverReason, classify_api_error
@@ -151,19 +152,59 @@ _SUMMARY_MISSING_CREDENTIAL_MARKERS: tuple[str, ...] = (
"no api key found",
)
_HYGIENE_IDLE_TIMEOUT_MARKERS: tuple[str, ...] = (
_HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS: tuple[str, ...] = (
"session hygiene compression timed out",
"hygiene compression deferred: turn-hold budget expired",
)
def _is_hygiene_idle_timeout_error(error: object) -> bool:
"""Return True when the durable cooldown came from a hygiene watchdog timeout.
def _is_hygiene_preagent_only_cooldown(error: object) -> bool:
"""Return True for a cooldown that belongs only to pre-agent hygiene.
That persist is intentional for the pre-agent hygiene pass (#74136) but
must not block the in-conversation compressor (#86972).
Hygiene watchdog timeouts and turn-hold deferrals intentionally persist
retry spacing for the pre-agent pass (#74136), but neither is evidence of
an auxiliary-model failure and neither may block the in-agent compressor
(#86972).
"""
text = str(error or "").strip().casefold()
return any(marker in text for marker in _HYGIENE_IDLE_TIMEOUT_MARKERS)
return any(
marker in text for marker in _HYGIENE_PREAGENT_ONLY_COOLDOWN_MARKERS
)
def _response_finish_reason(response: Any) -> str:
"""Return ``choices[0].finish_reason`` from a dict- or object-shaped response.
Mirrors the defensive message extraction in ``_generate_summary``: some
OpenAI-compatible proxies / local backends return plain dicts, others
return SDK objects, and either may omit the field entirely. Returns the
lowercased finish reason, or ``""`` when absent/unreadable.
"""
try:
if isinstance(response, dict):
choices = response.get("choices") or [{}]
first = choices[0] if choices else {}
reason = (
first.get("finish_reason")
if isinstance(first, dict)
else getattr(first, "finish_reason", None)
)
else:
choices = getattr(response, "choices", None) or []
reason = getattr(choices[0], "finish_reason", None) if choices else None
return str(reason).strip().lower() if reason else ""
except Exception:
return ""
# RuntimeError marker raised when the summarizer's generation stopped on the
# output-token cap (``finish_reason == "length"``). A length stop means the
# summary text is PARTIAL — persisting it as a compaction checkpoint would
# silently truncate the conversation's memory and feed the cut-off text back
# into every subsequent iterative-update prompt. The except-branch classifier
# below keys on this exact substring, so keep raise sites and the classifier
# in sync. (Ported from earendil-works/pi#7048 / commit 97fa14e39.)
_TRUNCATED_SUMMARY_MARKER = "finish_reason=length"
def _is_summary_access_or_quota_error(exc: Exception) -> bool:
@@ -252,6 +293,13 @@ COMPRESSED_SUMMARY_HAS_USER_TURN_KEY = "_compressed_summary_has_user_turn"
# rolling summary, so dropping or rewriting one destroys history.
MICRO_COMPACT_MARKER_KEY = "_micro_compact_marker"
_DB_PERSISTED_MARKER = "_db_persisted"
# Marks a message dict as carried-forward compaction tail (verbatim rows the
# compressor protected from summarization). archive_and_compact() archives
# these originals as rewind-style (active=0, compacted=0) instead of
# compacted=1, so they stop satisfying search_messages' recall filter and
# duplicating their live copies (#86366). Never persisted: _insert_message_rows
# only reads known columns.
_COMPACTION_TAIL_MARKER = "_compaction_tail"
PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY = "_proactive_prune_rearm_tokens"
_NO_USER_TASK_SENTINEL = "None. This session contains no user-authored turns."
@@ -1531,7 +1579,18 @@ def _estimate_msg_budget_tokens(msg: dict, charge_stale_thinking: bool = True) -
tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN
if not charge_stale_thinking:
return tokens
# The wire ships at most ONE of the generic thinking keys: every request
# build pops ``reasoning`` after (optionally) promoting it into
# ``reasoning_content`` (``apply_reasoning_content_policy``), and a
# non-empty stored ``reasoning_content`` always displaces it. Charging
# both keys double-counted the same thinking text on echo-back providers
# that persist it under both (#84371 comment: +53% vs real
# prompt_tokens). Mirror the wire: reasoning_content wins when present.
_rc = msg.get("reasoning_content")
_skip_reasoning_dup = isinstance(_rc, str) and bool(_rc.strip())
for key in _NEWEST_TURN_ONLY_BUDGET_KEYS:
if key == "reasoning" and _skip_reasoning_dup:
continue
tokens += _serialized_length_for_budget(msg.get(key)) // _CHARS_PER_TOKEN
# reasoning_details: charge only the thinking TEXT, never the signed /
# base64 envelope (#73298 second site; mirrors the preflight estimator's
@@ -2921,11 +2980,11 @@ class ContextCompressor(ContextEngine):
self._last_summary_error = None
return None
# Hygiene idle-watchdog timeouts persist the same column so the
# pre-agent pass can skip (#74136), but they are not evidence of a
# 429/aux-model fault. The in-conversation compressor has its own
# budget and must still be allowed to run (#86972).
if _is_hygiene_idle_timeout_error(state.get("error")):
# Hygiene watchdog timeouts and turn-hold deferrals persist the same
# column so the pre-agent pass can skip (#74136), but they are not
# evidence of a 429/aux-model fault. The in-conversation compressor has
# its own budget and must still be allowed to run (#86972).
if _is_hygiene_preagent_only_cooldown(state.get("error")):
# A later hygiene write can overwrite a previous aux-model row
# on the shared column. Drop any in-memory cooldown so the
# in-agent compressor is not still blocked after this refresh.
@@ -3262,10 +3321,20 @@ class ContextCompressor(ContextEngine):
equal the ENTIRE window — auto-compression can never fire because the
provider rejects the request before usage reaches 100% (#14690).
When the floor would meet or exceed the context window, trigger at
``_MIN_CTX_TRIGGER_RATIO`` (85%) of the window — high enough that a
small model uses most of its context before compacting, but below
100% so compaction fires before the provider rejects the request.
Near-minimum windows degenerate the same way without ever tripping an
equality check: at ``context_length == 65536`` the floored threshold
used to pass through at 64,000 — 97.7% of the window, ~1.5K tokens of
output room. Providers that silently truncate over-window prompts
instead of rejecting them (e.g. ollama's OpenAI-compatible endpoint)
never deliver the reactive context-overflow backstop either, so a
session rides into the window ceiling and every length-continuation
retry re-sends a window-filling prompt for a shrinking sliver of
output. Whenever the floor is the binding term, it is therefore capped
at ``_MIN_CTX_TRIGGER_RATIO`` (85%) of the effective input budget —
high enough that a small model uses most of its context before
compacting, but low enough that compaction fires while output room
remains. An explicit ``threshold_percent`` above 85% is user intent,
not the floor, and is not capped.
The provider reserves ``max_tokens`` of output space out of the same
window, so the usable INPUT budget is ``context_length - max_tokens``.
@@ -3281,13 +3350,17 @@ class ContextCompressor(ContextEngine):
effective_window = context_length
pct_value = int(effective_window * threshold_percent)
floored = max(pct_value, MINIMUM_CONTEXT_LENGTH)
# If flooring pushed the threshold to/over the effective window it can
# never be reached. Trigger at 85% of the effective input budget so a
# minimum-context model rides most of its budget before compacting
# instead of wasting half.
# The floor must not consume the window's output headroom: cap it at
# 85% of the effective input budget whenever it is the binding term.
# (An explicit threshold_percent above 85% is user intent — kept.)
trigger_cap = int(effective_window * ContextCompressor._MIN_CTX_TRIGGER_RATIO)
if effective_window > 0 and floored > pct_value and floored > trigger_cap:
floored = max(pct_value, trigger_cap)
# If the percentage itself reaches the effective window it can never
# be reached — trigger at 85% of the window, below 100% so compaction
# fires before the provider rejects (or silently clips) the request.
if effective_window > 0 and floored >= effective_window:
return max(1, min(int(effective_window * ContextCompressor._MIN_CTX_TRIGGER_RATIO),
effective_window - 1))
return max(1, min(trigger_cap, effective_window - 1))
return floored
def __init__(
self,
@@ -3522,6 +3595,14 @@ class ContextCompressor(ContextEngine):
# the session unchanged instead of destroying the middle window for a
# deterministic placeholder (#94448). Independent of abort_on_summary_failure.
self._last_summary_empty_content_failure: bool = False
# Set when summary generation ultimately fails because the summarizer
# stopped on its output-token cap (finish_reason == "length") — the
# summary text is PARTIAL and must never become a compaction
# checkpoint. compress() must ABORT and preserve the session unchanged
# exactly like the empty-content class: a truncated checkpoint
# silently destroys the compacted middle and compounds across
# iterative updates. (Ported from earendil-works/pi#7048.)
self._last_summary_truncated_failure: bool = False
# retrying on the main model, record the failure so gateway /
# CLI callers can still warn the user even though compression
# succeeded. Silent recovery would hide the broken config.
@@ -3601,6 +3682,26 @@ class ContextCompressor(ContextEngine):
self._verify_compaction_cleared_threshold = False
self.awaiting_real_usage_after_compression = False
def maybe_seed_preflight_display_tokens(self, preflight_tokens: int) -> None:
"""Seed ``last_prompt_tokens`` from a rough preflight estimate, display-only.
Policy (co-located with the rest of the speculative-seed lifecycle —
see ``snapshot_preflight_display_tokens`` /
``rollback_interrupted_preflight_display_tokens``): seed ONLY from
the 0 state ("no reading yet", #34282). Any non-zero value is
preserved — the -1 post-compression sentinel (#36718) AND any real
provider reading (#81481: the rough estimate intentionally
over-counts CJK / reasoning replay, 1.4-2.5x on heavy sessions, and
must never overwrite a real measurement).
Accepted trade-off: a provider reporting partial usage (e.g.
excluding cache-discounted tokens) pins the meter low until its next
report — preferred over estimator inflation.
"""
_last = self.last_prompt_tokens
if _last == 0 and preflight_tokens > _last:
self.last_prompt_tokens = preflight_tokens
def snapshot_preflight_display_tokens(self) -> int:
"""Capture the display token count before a speculative preflight seed."""
return self.last_prompt_tokens
@@ -3984,11 +4085,16 @@ class ContextCompressor(ContextEngine):
# Same newest-turn-only thinking charge as the tail-cut walk
# (#73624) — this boundary decides which tool results stay
# prunable, and overcharging stale thinking shrinks that window.
# Echo-back routes charge every turn (#84371 estimator parity).
_newest_asst_idx = _last_assistant_index(result)
_charge_all_thinking = self._stale_thinking_on_wire()
for i in range(len(result) - 1, -1, -1):
msg = result[i]
msg_tokens = _estimate_msg_budget_tokens(
msg, charge_stale_thinking=(i == _newest_asst_idx)
msg,
charge_stale_thinking=(
_charge_all_thinking or i == _newest_asst_idx
),
)
if accumulated + msg_tokens > protect_tail_tokens and (len(result) - i) >= min_protect:
boundary = i
@@ -5241,22 +5347,13 @@ This compaction should PRIORITISE preserving all information related to the focu
)
if self._compression_cancelled():
raise AuxiliaryExplicitCancellation()
# ``_validate_llm_response`` only guarantees ``choices[0].message``
# exists, not that it's an object with ``.content``. Some
# OpenAI-compatible proxies / local backends return a dict- or
# str-shaped message; coerce defensively instead of crashing.
if isinstance(response, dict):
choices = response.get("choices") or [{}]
message = choices[0].get("message") if isinstance(choices[0], dict) else getattr(choices[0], "message", None)
else:
message = response.choices[0].message
if isinstance(message, dict):
content = message.get("content")
else:
content = getattr(message, "content", message)
# Handle cases where content is not a string (e.g., dict from llama.cpp)
if not isinstance(content, str):
content = str(content) if content else ""
# Dict/object/str messages + reasoning-field fallback (DeepSeek /
# Qwen / Kimi return content="" with the summary in
# reasoning_content). Cap the fallback so a CoT dump cannot
# become the compaction summary.
content = extract_content_or_reasoning(
response, max_reasoning_chars=8000
)
# Some OpenAI-compatible proxies (e.g. cmkey.cn, one-api channels)
# return a well-formed HTTP 200 with an empty or whitespace-only
# ``content`` instead of an error or empty ``choices``. That payload
@@ -5273,6 +5370,23 @@ This compaction should PRIORITISE preserving all information related to the focu
f"(provider={self.provider or 'auto'} "
f"model={self.summary_model or self.model})"
)
# A finish_reason of "length" means the summarizer hit its output
# token cap mid-generation: the text present is PARTIAL. Persisting
# a partial summary as the compaction checkpoint silently truncates
# the conversation's memory — the cut-off text replaces the real
# middle turns AND is fed back into every subsequent iterative
# update prompt, compounding the loss across compactions. Treat it
# as a failure so it routes through the same main-model fallback +
# abort machinery as other degraded responses instead of becoming
# a checkpoint. (Ported from earendil-works/pi#7048.)
if _response_finish_reason(response) == "length":
raise RuntimeError(
"Context compression summary was truncated "
f"({_TRUNCATED_SUMMARY_MARKER}): generation hit the output "
"token cap and the summary is incomplete "
f"(provider={self.provider or 'auto'} "
f"model={self.summary_model or self.model})"
)
# Strip reasoning blocks the summarizer model may have emitted
# (<think>...</think> etc. from thinking models like MiniMax,
# DeepSeek, QwQ). Without this the trace is stored in
@@ -5300,6 +5414,7 @@ This compaction should PRIORITISE preserving all information related to the focu
self._last_summary_auth_failure = False
self._last_summary_network_failure = False
self._last_summary_empty_content_failure = False
self._last_summary_truncated_failure = False
return self._with_summary_prefix(summary)
except Exception as e:
# ``call_llm`` raises ``RuntimeError`` for two very different cases:
@@ -5373,6 +5488,16 @@ This compaction should PRIORITISE preserving all information related to the focu
or "llm returned none response" in _err_str
or "llm returned invalid response" in _err_str
)
# Summarizer stopped on its output-token cap (finish_reason ==
# "length"): the summary text is partial and must never become a
# compaction checkpoint. Same degraded-response handling shape as
# empty content — one main-model retry (a larger/unconstrained
# model may finish the summary), then ABORT preserving the session
# unchanged. (Ported from earendil-works/pi#7048.)
_is_truncated_summary = (
isinstance(e, RuntimeError)
and _TRUNCATED_SUMMARY_MARKER in _err_str
)
# Authentication, permission, and exhausted-quota failures are NOT
# transient or fixable by retrying the same request. Flag them so
# compress() preserves the session instead of rotating into a
@@ -5398,13 +5523,15 @@ This compaction should PRIORITISE preserving all information related to the focu
e,
)
if (
(_is_model_not_found or _is_timeout or _is_json_decode or _is_streaming_closed or _is_empty_content)
(_is_model_not_found or _is_timeout or _is_json_decode or _is_streaming_closed or _is_empty_content or _is_truncated_summary)
and self.summary_model
and self.summary_model != self.model
and not getattr(self, "_summary_model_fallen_back", False)
):
if _is_json_decode:
_reason = "returned invalid JSON"
elif _is_truncated_summary:
_reason = "returned a truncated summary (output token cap)"
elif _is_empty_content:
_reason = "returned empty content"
elif _is_model_not_found:
@@ -5464,7 +5591,7 @@ This compaction should PRIORITISE preserving all information related to the focu
min(self._consecutive_timeout_failures,
len(_TIMEOUT_COOLDOWN_LADDER)) - 1
]
elif _is_json_decode or _is_streaming_closed or _is_empty_content:
elif _is_json_decode or _is_streaming_closed or _is_empty_content or _is_truncated_summary:
_transient_cooldown = 30
else:
_transient_cooldown = 60
@@ -5483,6 +5610,8 @@ This compaction should PRIORITISE preserving all information related to the focu
# auth-failure carve-out; independent of abort_on_summary_failure.
if _is_streaming_closed:
self._last_summary_network_failure = True
elif _is_truncated_summary:
self._last_summary_truncated_failure = True
elif _is_empty_content:
self._last_summary_empty_content_failure = True
logger.warning(
@@ -6553,6 +6682,30 @@ This compaction should PRIORITISE preserving all information related to the focu
idx += 1
return idx
def _stale_thinking_on_wire(self) -> bool:
"""Whether the active route replays stale thinking text (#84371).
The tail-budget walks and the preflight trigger must charge the SAME
stale-thinking policy or a reasoning-heavy session can look
over-threshold to one and fully tail-protected to the other — the
infinite ineffective compaction loop. Echo-back chat-completions
families (DeepSeek/Kimi/MiMo thinking mode) replay stored
``reasoning_content`` on EVERY assistant turn, so the walk must
charge it everywhere; codex_responses and strict providers never
ship the text keys, so newest-turn-only stands (#73624).
"""
try:
from agent.message_sanitization import stale_thinking_reaches_wire
return stale_thinking_reaches_wire(
getattr(self, "api_mode", "") or "",
getattr(self, "provider", "") or "",
getattr(self, "model", "") or "",
getattr(self, "base_url", "") or "",
)
except Exception:
return False
def _find_tail_cut_by_tokens(
self, messages: List[Dict[str, Any]], head_end: int,
token_budget: int | None = None,
@@ -6598,12 +6751,20 @@ This compaction should PRIORITISE preserving all information related to the focu
# fields any transport still replays (#73624) — every older turn's
# reasoning/reasoning_content is stripped or padded at send time,
# so charging it here spends tail budget on bytes that never ship.
# Exception: echo-back providers (DeepSeek/Kimi/MiMo thinking mode
# on chat_completions) replay stale thinking on EVERY turn — charge
# it everywhere so this walk agrees with the preflight trigger
# (#84371 estimator parity).
_newest_asst_idx = _last_assistant_index(messages)
_charge_all_thinking = self._stale_thinking_on_wire()
for i in range(n - 1, head_end - 1, -1):
msg = messages[i]
msg_tokens = _estimate_msg_budget_tokens(
msg, charge_stale_thinking=(i == _newest_asst_idx)
msg,
charge_stale_thinking=(
_charge_all_thinking or i == _newest_asst_idx
),
)
# Stop once we exceed the soft ceiling (unless we haven't hit min_tail yet)
if accumulated + msg_tokens > soft_ceiling and (n - i) >= min_tail:
@@ -6631,7 +6792,10 @@ This compaction should PRIORITISE preserving all information related to the focu
for j in range(n - 1, head_end - 1, -1):
raw_msg = messages[j]
raw_tok = _estimate_msg_budget_tokens(
raw_msg, charge_stale_thinking=(j == _newest_asst_idx)
raw_msg,
charge_stale_thinking=(
_charge_all_thinking or j == _newest_asst_idx
),
)
if raw_accumulated + raw_tok > raw_budget and (n - j) >= min_tail:
cut_idx = j
@@ -6945,6 +7109,18 @@ This compaction should PRIORITISE preserving all information related to the focu
logger.info("micro-summarization call failed: %s", exc)
return None
# A length stop means the merged rolling summary is partial —
# persisting it would silently drop the tail of the merge and feed
# the cut-off text into every later micro-compact pass. Leave the
# exchange unabsorbed instead; a later pass retries it.
# (Same class as _generate_summary's guard; pi#7048.)
if _response_finish_reason(response) == "length":
logger.warning(
"micro-summarization output hit the token cap "
"(finish_reason=length) — discarding partial summary",
)
return None
message = response.choices[0].message
if isinstance(message, dict):
content = message.get("content")
@@ -7317,7 +7493,15 @@ This compaction should PRIORITISE preserving all information related to the focu
if not session_db or not session_id:
return
try:
session_db.archive_and_compact(session_id, compacted_messages)
# The splice result is [..verbatim prefix.., summary_marker,
# ..verbatim suffix..]: every row except the single marker is a
# carried-forward original (#86366) — archive their pre-splice
# originals rewind-style instead of compacted=1.
session_db.archive_and_compact(
session_id,
compacted_messages,
tail_count=max(0, len(compacted_messages) - 1),
)
# Shared post-commit contract with the in-place batch commit and
# the proactive prune (#98450) — one stamp site for the class.
stamp_db_persisted_markers(compacted_messages)
@@ -7527,7 +7711,8 @@ This compaction should PRIORITISE preserving all information related to the focu
self._last_compress_refused_would_grow = False
self._last_compression_made_progress = False
# NOTE: do NOT reset _last_summary_auth_failure,
# _last_summary_network_failure, or _last_summary_empty_content_failure
# _last_summary_network_failure, _last_summary_empty_content_failure,
# or _last_summary_truncated_failure
# here. These flags are set by _generate_summary() on a terminal
# failure and are already cleared on a successful summary. Resetting them eagerly defeats the cooldown
# protection: _generate_summary() returns None from the cooldown
@@ -7884,6 +8069,7 @@ This compaction should PRIORITISE preserving all information related to the focu
or self._last_summary_auth_failure
or self._last_summary_network_failure
or self._last_summary_empty_content_failure
or self._last_summary_truncated_failure
):
n_skipped = compress_end - compress_start
self._last_summary_dropped_count = 0 # nothing actually dropped
@@ -7893,6 +8079,8 @@ This compaction should PRIORITISE preserving all information related to the focu
telemetry["failure_class"] = "summary_auth_failure"
elif self._last_summary_network_failure:
telemetry["failure_class"] = "summary_network_failure"
elif self._last_summary_truncated_failure:
telemetry["failure_class"] = "summary_truncated_failure"
elif self._last_summary_empty_content_failure:
telemetry["failure_class"] = "summary_empty_content_failure"
else:
@@ -7922,6 +8110,16 @@ This compaction should PRIORITISE preserving all information related to the focu
"recovers, or continue the conversation as-is.",
n_skipped,
)
elif self._last_summary_truncated_failure:
logger.warning(
"Summary generation failed (output hit the token cap; "
"summary is incomplete) — aborting compression. "
"%d message(s) preserved unchanged; the session was NOT "
"rotated. A truncated summary would silently lose "
"context: retry with /compress, or raise the "
"summarizer's output budget.",
n_skipped,
)
elif self._last_summary_empty_content_failure:
logger.warning(
"Summary generation failed (LLM returned empty content) — "
@@ -8172,6 +8370,11 @@ This compaction should PRIORITISE preserving all information related to the focu
if _force_user_leading and first_tail_visible_idx is not None:
_merge_target_idx = first_tail_visible_idx
for tail_idx, msg in enumerate(tail_messages):
# Tag the carried-forward tail so archive_and_compact() can
# classify these rows' originals as superseded-duplicate
# (rewind semantics) instead of "summarized away" (#86366).
if isinstance(msg, dict):
msg[_COMPACTION_TAIL_MARKER] = True
if _merge_summary_into_tail and tail_idx == _merge_target_idx:
# Merge the summary into the tail message that collided.
old_content = msg.get("content", "")
+258 -3
View File
@@ -92,11 +92,19 @@ _TERMINAL_COMPRESSION_PROVENANCES = frozenset(
}
)
# Cooldown armed when a compression SPLIT fails (session_split_failed /
# rotation rollback, #97948 symptom B). Deliberately the FIRST rung of the
# timeout ladder (60/300/900 in context_compressor.py), not the 600s
# _SUMMARY_FAILURE_COOLDOWN_SECONDS: a split failure is usually a transient
# lease/DB condition, unlike a persistent summary-provider fault.
_SPLIT_FAILURE_COOLDOWN_SECONDS = 60
# Stable marker the gateway matches on to re-tag the auto-compaction lifecycle
# status as ``kind="compacting"`` (tui_gateway/server.py::_status_update), so
# drivers like the desktop app can show an explicit "Summarizing…" indicator
# instead of the transcript appearing to silently reset. Keep the marker phrase
# intact if you reword COMPACTION_STATUS.
# intact if you reword COMPACTION_STATUS. Idle/preflight/retry lines do not
# contain this marker — ``is_compaction_progress_status`` covers those too.
COMPACTION_STATUS_MARKER = "Compacting context"
COMPACTION_STATUS = (
f"🗜️ {COMPACTION_STATUS_MARKER} — summarizing earlier conversation so I can continue..."
@@ -206,6 +214,40 @@ ROUTINE_COMPRESSION_STATUS_SAMPLES = (
)
def is_compaction_progress_status(text: str | None) -> bool:
"""True for in-progress auto-compaction lifecycle lines (not the done edge).
``tui_gateway.server._status_update`` re-tags matching ``lifecycle``
statuses as ``kind="compacting"`` so TUI and desktop can show a summarizing
indicator for the whole pause. Matching only ``COMPACTION_STATUS_MARKER``
left idle/preflight/retry lines looking like a hung turn (#97239).
The terminal ``COMPACTION_DONE_STATUS`` is emitted as ``kind="compacted"``
and must not match here.
"""
if not isinstance(text, str):
return False
body = text.strip()
if not body:
return False
if COMPACTION_STATUS_MARKER in body:
return True
if body == COMPACTION_DONE_STATUS:
return False
lowered = body.lower()
if "compaction complete" in lowered:
return False
# Failure-class overflow warning mentions compression but is a blocked
# notice, not progress — keep it lifecycle so chat gateways stay loud.
if "compression is currently blocked" in lowered:
return False
return (
"compact" in lowered
or "compress" in lowered
or "context reduced to" in lowered
)
def _builtin_memory_prompt_snapshot(agent: Any) -> Optional[Tuple[str, str]]:
"""Return the built-in memory text that can affect a system prompt.
@@ -320,6 +362,7 @@ _COMPRESSOR_ATTEMPT_STATE_FIELDS = (
"_last_summary_auth_failure",
"_last_summary_network_failure",
"_last_summary_empty_content_failure",
"_last_summary_truncated_failure",
"_last_aux_model_failure_error",
"_last_aux_model_failure_model",
"_summary_model_fallen_back",
@@ -1856,6 +1899,76 @@ def compression_skipped_due_to_lock(agent: Any) -> bool:
return _sig is True or isinstance(_sig, str)
def _get_context_compression_timeout_state(
agent: Any,
*,
create: bool,
) -> Optional[Tuple[Any, Optional[threading.local]]]:
"""Return the stable lock and thread-local timeout state for an agent."""
try:
attributes = vars(agent)
except TypeError:
return None
lock = attributes.setdefault(
"_context_compression_timeout_state_lock",
threading.Lock(),
)
with lock:
state = attributes.get("_context_compression_timeout_state")
if create and not isinstance(state, threading.local):
state = threading.local()
attributes["_context_compression_timeout_state"] = state
return lock, state if isinstance(state, threading.local) else None
def reset_context_compression_timeout_outcome(agent: Any) -> None:
"""Clear the current thread's owned-compression timeout outcome.
The compatibility mirror ``agent._last_compression_timed_out`` is the
simple per-attempt flag landed by #98424 (turn-start preflight fail-closed
boundary); it stays authoritative for minimal/older agent doubles that do
not support ``vars()``.
"""
locked_state = _get_context_compression_timeout_state(agent, create=True)
if locked_state is None or locked_state[1] is None:
agent._last_compression_timed_out = False
return
lock, state = locked_state
with lock:
state.timed_out = False
agent._last_compression_timed_out = False
def mark_context_compression_timed_out(agent: Any) -> None:
"""Mark the current owned compression as host-timed-out."""
locked_state = _get_context_compression_timeout_state(agent, create=True)
if locked_state is None or locked_state[1] is None:
agent._last_compression_timed_out = True
return
lock, state = locked_state
with lock:
state.timed_out = True
agent._last_compression_timed_out = True
def context_compression_timed_out(agent: Any) -> bool:
"""Return whether this thread's owned compression hit its host timeout.
A single agent may receive overlapping automatic/manual entrypoints. The
thread-local outcome prevents one entrypoint's reset from hiding another's
timeout. The attribute fallback supports older/minimal agent doubles.
Every read is type-pinned to avoid MagicMock auto-attributes.
"""
locked_state = _get_context_compression_timeout_state(agent, create=False)
if locked_state is not None:
lock, state = locked_state
with lock:
if isinstance(state, threading.local):
return getattr(state, "timed_out", None) is True
return getattr(agent, "_last_compression_timed_out", None) is True
def compression_blocked_transiently(agent: Any) -> bool:
"""Type-pinned read of the transient-block signal (#97488).
@@ -3897,8 +4010,27 @@ def compress_context(
)
# Incoming-message interrupts and active-turn redirects must not tear an
# atomic summary in half (#23975). Explicit stop surfaces set a separate
# Event atomically; never infer cause from the racy message fields.
# Event atomically; never infer cause from the racy message fields. A
# host timeout also cancels the attempt's commit fence. Feed BOTH into
# the protected auxiliary-call seam so the compression owner unwinds
# promptly while an isolated provider stream finishes or closes in its
# daemon worker. Otherwise four timed-out streams retain all four shared
# compression-pool slots until the auxiliary stream's longer absolute
# ceiling expires.
_hard_cancel_event = getattr(agent, "_hard_interrupt_requested", None)
def _compression_cancel_requested() -> bool:
return bool(
(
_hard_cancel_event is not None
and _hard_cancel_event.is_set()
)
or (
commit_fence is not None
and commit_fence.is_cancelled
)
)
try:
# F6: never start expensive summary work for an already-cancelled
# fence (a stale queued job admitted after host departure).
@@ -3911,7 +4043,7 @@ def compress_context(
compressed = messages
else:
with aux_progress_hook(_progress_hook), aux_interrupt_protection(
cancel_event=_hard_cancel_event
cancel_check=_compression_cancel_requested
):
compressed = compress_fn(messages, **compress_kwargs)
# Freeze a hard stop that arrived after the final provider
@@ -4075,6 +4207,26 @@ def compress_context(
"Compression made no progress (session=%s) — skipping boundary rewrite.",
agent.session_id or "none",
)
# Dead-loop breaker (#84371): a fired compaction that returns the
# transcript UNCHANGED will fail identically next turn unless the
# transcript changes — yet this path recorded telemetry only, so
# auto-compress re-fired every turn, each attempt burning a full
# aux summarization (6+/10min in the wild). Arm the transient
# structural backoff so the next attempts are deferred; any
# successful boundary lifts it, and manual /compress overrides it.
try:
_no_progress_recorder = getattr(
agent.context_compressor, "_record_structural_no_op", None
)
if callable(_no_progress_recorder):
_no_progress_recorder(
"compaction returned the transcript unchanged "
"(no_progress)"
)
except Exception:
logger.debug(
"no-progress backoff arm failed", exc_info=True
)
_existing_sp = getattr(agent, "_cached_system_prompt", None)
if not _existing_sp:
_existing_sp = agent._build_system_prompt(system_message)
@@ -4400,6 +4552,21 @@ def compress_context(
# away regardless of whether the id rotates).
agent.commit_memory_session(messages)
# Pop the #86366 carried-forward-tail tags BEFORE the size
# estimate and BEFORE any non-in-place path can see them:
# the private `_compaction_tail` key must not inflate the
# anti-growth token estimate (it tipped break-even
# transcripts into a false "would grow" refusal) and must
# not ride rotation handoffs into the provider payload.
# Remember the tagged dicts by identity — the salvage pass
# below may swap `compressed` for a subset, and tail_count
# must describe the FINAL committed list.
_tail_tagged_ids = {
id(m)
for m in compressed
if isinstance(m, dict) and m.pop("_compaction_tail", None)
}
# Anti-growth guard at the COMMIT SITE: never persist a
# compression that makes the transcript larger (observed:
# 379K -> 687K when the generated summary plus retained
@@ -4532,6 +4699,15 @@ def compress_context(
PROACTIVE_PRUNE_REARM_MODEL_CONFIG_KEY,
)
# Tail rows carried the _compaction_tail tag from
# compress() (#86366; popped above, before the size
# estimate): their originals must be archived as
# superseded duplicates (rewind-style), not compacted=1.
# Count against the FINAL list — salvage may have
# dropped rows.
_tail_count = sum(
1 for m in compressed if id(m) in _tail_tagged_ids
)
agent._session_db.archive_and_compact(
agent.session_id,
compressed,
@@ -4540,6 +4716,7 @@ def compress_context(
},
watermark=_commit_watermark,
lock_holder=_lock_holder,
tail_count=_tail_count,
)
split_status = "in_place_committed"
# Post-commit contract (#98450, mirrors
@@ -4620,13 +4797,27 @@ def compress_context(
# before. Deliberately NOT extended to the compression lease:
# a lease is re-acquirable, so a transient miss here would
# abort a rotation that would otherwise have committed.
#
# AUTOMATIC stamps (tui_shutdown, ws_disconnect, orphan
# reap, idle/LRU evict — is_automatic_end_reason) do NOT
# trip this guard: publish_compression_child treats them
# as stale-by-construction and clears them in its own
# transaction (#88197 Bug 1), so aborting here would keep
# rotation wedged on exactly the stamp the publish can
# heal. Only deliberate boundaries (compression,
# session_reset, explicit close) abort before the flush.
_parent_row_reader = getattr(agent._session_db, "get_session", None)
_parent_already_ended = False
if callable(_parent_row_reader):
try:
from hermes_state_common import is_automatic_end_reason
_parent_row = _parent_row_reader(old_session_id) or {}
_parent_already_ended = (
_parent_row.get("ended_at") is not None
and not is_automatic_end_reason(
_parent_row.get("end_reason")
)
)
except Exception:
# Fail OPEN: an unreadable row must not turn a cheap
@@ -4694,6 +4885,8 @@ def compress_context(
profile_name=_profile_for_child,
compression_lock_holder=_lock_holder,
require_compression_lease=_lock_holder is not None,
require_lease_refresh=_lock_holder is not None,
lease_ttl_seconds=_lock_ttl,
watermark=(
_commit_watermark
if _foreign_tail_ceiling is not None
@@ -4955,6 +5148,51 @@ def compress_context(
"_proactive_prune_rearm_tokens"
]
)
elif (
in_place
and split_status != "in_place_committed"
and messages_before_compression is not None
):
# In-place sibling of the rotation rollback above (#99477).
# archive_and_compact() is atomic, so a raise before it
# returned means EVERY pre-compaction row is still
# ``active = 1`` in state.db — nothing was archived and the
# compacted set was never inserted. But ``compressed`` is
# the marker-swept output of compress()
# (_strip_persistence_markers, #57491) and the post-commit
# ``stamp_db_persisted_markers`` never ran, so handing it
# back makes the next append-only flush treat the whole
# compacted transcript as new and INSERT it ON TOP of the
# rows it was supposed to replace. The active set then holds
# the summary AND the turns it summarized; the next resume
# reloads both, the token count goes UP, preflight fires
# again, and each failed attempt appends another copy of the
# protected head + tail (#99477: ~15 real turns stored as
# 3,814 rows, the first user message repeated 893 times).
#
# Gate on ``split_status`` rather than ``compacted_in_place``:
# it is assigned on the statement immediately after the
# atomic commit returns, so a committed compaction can never
# be rolled back into a live/durable mismatch of the
# opposite sign.
#
# The deepcopy carries each row's _DB_PERSISTED_MARKER from
# the pre-compression snapshot, so the restored transcript is
# correctly skipped by the flush, and replacing every dict
# breaks _db_flush_scan_prefix identity (same reasoning as
# the rotation branch — no explicit clear needed).
messages[:] = copy.deepcopy(messages_before_compression)
compressed = messages
_compression_made_progress = False
# Runway rolls back with the transcript, exactly as above:
# compress() zeroed it in memory, and the durable clear only
# rides the archive_and_compact that just failed.
if "_proactive_prune_rearm_tokens" in _compressor_attempt_snapshot:
agent.context_compressor._proactive_prune_rearm_tokens = (
_compressor_attempt_snapshot[
"_proactive_prune_rearm_tokens"
]
)
split_status = (
"aborted"
if locals().get("old_session_id") is None and not in_place
@@ -4972,6 +5210,21 @@ def compress_context(
)
else:
logger.warning("Session DB compression split failed — new session will NOT be indexed: %s", e)
# Arm the failure cooldown so the next turn cannot immediately
# re-run the identical doomed compression (#97948 symptom B).
# try/except mirrors the sibling record_rejected_compaction
# call above: this runs inside the split-failure handler, and
# a stub compressor must not mask the original error.
try:
agent.context_compressor._record_compression_failure_cooldown(
_SPLIT_FAILURE_COOLDOWN_SECONDS,
f"session_split_failed: {e}",
)
except Exception:
logger.debug(
"could not record split-failure cooldown",
exc_info=True,
)
# Compaction-boundary bookkeeping, computed once. `old_session_id` is only
# bound in the rotation branch; in-place leaves it unset. `_boundary_parent`
@@ -5102,6 +5355,7 @@ def compress_context(
# the next response with usage re-anchors (its structural id/index
# check would also fail closed, but explicit is safer).
agent._usage_anchor = None
agent._turn_base_usage_anchor = None
# Arm the effectiveness verdict only after a completed rewrite crosses
# the full compaction boundary. Exceptions, aborts, and no-op attempts
# leave this false, so unrelated later usage cannot be charged to an
@@ -5661,6 +5915,7 @@ __all__ = [
"COMPACTION_STATUS",
"COMPACTION_DONE_STATUS",
"COMPACTION_STATUS_MARKER",
"is_compaction_progress_status",
"check_compression_model_feasibility",
"replay_compression_warning",
"compress_context",
+393 -48
View File
@@ -35,6 +35,7 @@ from agent.conversation_compression import (
PRE_API_COMPRESSION_STATUS_TEMPLATE,
compression_blocked_transiently,
compression_skipped_due_to_lock,
context_compression_timed_out,
conversation_history_after_compression,
)
from agent.context_engine import automatic_compaction_status_message
@@ -42,6 +43,7 @@ from agent.display import KawaiiSpinner
from agent.error_classifier import FailoverReason, classify_api_error
from agent.message_metadata import append_message
from agent.turn_context import (
PreflightCompressionTimedOut,
_compression_warrants_another_preflight_pass,
_review_fork_first_request_pending,
build_turn_context,
@@ -297,6 +299,16 @@ _HANDOFF_SKIP_FINAL_RESPONSE = (
"awaiting your next message."
)
# Terminal final_response for a turn ended because context compression hit its
# host progress-aware timeout while the request was still oversized (#98722,
# salvaged from #98741). Sending the unchanged request would only bounce off
# the provider's overflow error and re-enter compression in the same turn.
_COMPRESSION_TIMEOUT_FINAL_RESPONSE = (
"Context compression timed out without reducing this conversation. "
"No messages were dropped. Start a fresh session with /new, or check "
"auxiliary.compression before retrying /compress."
)
# Stable prefix of the local interrupt status string emitted when a turn is
# cancelled while waiting on the provider. Surfaces (ACP, TUI) match on this
@@ -640,6 +652,40 @@ def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str
)
def _maybe_grow_local_window(agent: Any, compressor: Any,
request_tokens: int) -> Optional[int]:
"""Try growing the managed local model's context window before
compressing. Returns the new window when the ladder granted one, else
None (hold / at native / not a managed local session).
The window ladder's design order: models launch at their zero-spill
window and grow toward native max as the session needs room;
compression is the move of last resort. Cheap for every non-local
provider: one lowercase compare, no imports.
"""
provider = (getattr(agent, "provider", "") or "").strip().lower()
if provider not in ("llamacpp", "llama.cpp", "llama-cpp", "custom"):
return None
base_url = getattr(agent, "base_url", "") or ""
if "127.0.0.1" not in base_url and "localhost" not in base_url:
return None
try:
from hermes_cli.local_runtime.growth import maybe_grow_window
current_window = int(getattr(compressor, "context_length", 0) or 0)
if current_window <= 0:
return None
return maybe_grow_window(
getattr(agent, "model", "") or "",
base_url=base_url,
session_tokens=int(request_tokens),
current_window=current_window,
)
except Exception as exc: # noqa: BLE001 — growth must never break a turn
logger.debug("local window growth check failed: %s", exc)
return None
def _ra():
"""Lazy reference to ``run_agent`` so callers can patch
``run_agent.handle_function_call`` / ``run_agent._set_interrupt`` /
@@ -1578,6 +1624,41 @@ def _compression_deferred_result(
}
def _provider_overflow_exhausted_result(
agent,
messages: List[Dict],
conversation_history,
api_call_count: int,
request_pressure_tokens: int,
max_compression_attempts: int,
) -> Dict[str, Any]:
"""Fail closed when a rebuilt request is still too large after recovery."""
agent._flush_status_buffer()
logger.error(
"%sContext compression failed after %d attempts; rebuilt request "
"remains over threshold at ~%s tokens.",
agent.log_prefix,
max_compression_attempts,
f"{request_pressure_tokens:,}",
)
agent._persist_session(messages, conversation_history)
final_response = (
"Context length exceeded: compression could not reduce the rebuilt "
"request below the safe threshold."
)
return {
"final_response": final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
"turn_exit_reason": "context_compression_exhausted",
}
def _rewrite_system_content_blocks(system_message: dict, effective: str) -> bool:
"""Rewrite a cache-decorated system message in place, keeping its blocks.
@@ -1907,6 +1988,7 @@ def run_conversation(
persist_user_timestamp: Optional[float] = None,
persist_user_display_kind: Optional[str] = None,
persist_user_display_metadata: Optional[Dict[str, Any]] = None,
persist_user_platform_id: Optional[str] = None,
moa_config: Optional[dict[str, Any]] = None,
) -> Dict[str, Any]:
"""
@@ -1932,6 +2014,10 @@ def run_conversation(
the message unchanged.
persist_user_display_metadata: Optional payload for that event
(e.g. a delegation's task count).
persist_user_platform_id: Optional platform-side message id (e.g. the
Discord/Telegram message id) to store as metadata on that
persisted user message, so restart drain-window recovery can
dedup an interrupted turn against the transcript.
or queuing follow-up prefetch work.
Returns:
@@ -1973,28 +2059,73 @@ def run_conversation(
# ``build_turn_context``. It mutates ``agent`` exactly as the inline code
# did and returns the locals the loop below reads back. See
# ``agent/turn_context.py``.
_ctx = build_turn_context(
agent,
user_message,
system_message,
conversation_history,
task_id,
stream_callback,
persist_user_message,
persist_user_timestamp,
persist_user_display_kind=persist_user_display_kind,
persist_user_display_metadata=persist_user_display_metadata,
restore_or_build_system_prompt=_restore_or_build_system_prompt,
install_safe_stdio=_install_safe_stdio,
sanitize_surrogates=_sanitize_surrogates,
summarize_user_message_for_log=_summarize_user_message_for_log,
set_session_context=set_session_context,
set_current_write_origin=set_current_write_origin,
ra=_ra,
# MoA turns append per-call aggregated context to the API copy of the
# user message, so no byte-stable api_content sidecar can be stamped.
moa_active=bool(moa_config),
)
try:
_ctx = build_turn_context(
agent,
user_message,
system_message,
conversation_history,
task_id,
stream_callback,
persist_user_message,
persist_user_timestamp,
persist_user_display_kind=persist_user_display_kind,
persist_user_display_metadata=persist_user_display_metadata,
persist_user_platform_id=persist_user_platform_id,
restore_or_build_system_prompt=_restore_or_build_system_prompt,
install_safe_stdio=_install_safe_stdio,
sanitize_surrogates=_sanitize_surrogates,
summarize_user_message_for_log=_summarize_user_message_for_log,
set_session_context=set_session_context,
set_current_write_origin=set_current_write_origin,
ra=_ra,
# MoA turns append per-call aggregated context to the API copy of the
# user message, so no byte-stable api_content sidecar can be stamped.
moa_active=bool(moa_config),
)
except PreflightCompressionTimedOut as _preflight_timeout_exc:
# Turn-start fail-closed boundary (#98424): preflight compression hit
# the host's progress-aware timeout while the request was still
# oversized, so no provider call was sent. Convert the typed exception
# into the same typed recovery result the in-loop consumers return
# (salvaged #98741 / PR #99710) instead of letting it escape to the
# surfaces' generic exception handlers — the gateway deliberately
# hides raw exception text from users, which would bury the
# actionable "run /compress and retry" guidance and skip the
# compression_exhausted clean-session recovery contract.
logger.warning(
"Turn-start preflight compression timed out — ending turn with "
"typed recovery result: %s",
_preflight_timeout_exc,
)
# build_turn_context registered this turn's in-flight tripwire slot
# (note_turn_start) but the early return skips the persist funnel
# that normally clears it — clear it here so the next turn does not
# log a spurious "concurrent turns on one session" warning. The
# inbound user row is intentionally NOT persisted on this path: the
# gateway skips transcript persistence for compression_exhausted
# results to prevent the session-growth loop (#7100), and the
# auto-reset moves future input to a clean session.
from agent.agent_runtime_helpers import note_turn_persisted
note_turn_persisted(agent)
# Intentionally NOT _COMPRESSION_TIMEOUT_FINAL_RESPONSE: the boundary's
# exception text carries per-request context (token count, "provider
# call was not sent") that is the actionable guidance this handler
# exists to surface; the in-loop constant describes a different state
# (compression ran and could not reduce).
_final_response = str(_preflight_timeout_exc)
return {
"final_response": _final_response,
"messages": list(conversation_history or []),
"completed": False,
"api_calls": 0,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
"turn_exit_reason": "context_compression_timeout",
}
user_message = _ctx.user_message
original_user_message = _ctx.original_user_message
messages = _ctx.messages
@@ -2046,6 +2177,17 @@ def run_conversation(
max_compression_attempts = getattr(agent, "max_compression_attempts", 3)
_last_preflight_pressure: Optional[int] = None
_preflight_compression_blocked = _ctx.preflight_compression_blocked
# A provider overflow is stronger evidence than the rough-estimate
# calibration that normally defers preflight immediately after compaction.
# Keep recovery armed until the rebuilt, complete request is below the
# configured compression threshold. Without this handoff, a compaction
# that drops rows but grows the actual prompt can be sent straight back to
# the provider while awaiting_real_usage_after_compression is true.
_provider_overflow_recovery_pending = False
# Armed when a compression host-timeout terminates the turn (#98722,
# salvaged from #98741); finalize below reuses the gateway's existing
# context-recovery contract (error/partial/compression_exhausted).
_compression_timeout_exhausted = False
_turn_exit_reason = "unknown" # Diagnostic: why the loop ended
# Last composed answer intentionally held back by a verification gate. If
# that continuation consumes the remaining budget, this is the best
@@ -2302,7 +2444,10 @@ def run_conversation(
# repair_message_sequence_with_cursor also recomputes the SessionDB
# flush cursor (_last_flushed_db_idx) when repair compacts the list,
# so the turn-end flush doesn't skip the assistant/tool chain (#44837).
from agent.agent_runtime_helpers import repair_message_sequence_with_cursor
from agent.agent_runtime_helpers import (
fill_empty_non_final_wire_payload,
repair_message_sequence_with_cursor,
)
repaired_seq = repair_message_sequence_with_cursor(agent, messages)
if repaired_seq > 0:
request_logger.info(
@@ -2332,29 +2477,9 @@ def run_conversation(
# from every outgoing copy so strict OpenAI-compatible backends
# don't reject the request after a model switch or resumed typed
# event row enters the live history.
_display_kind = api_msg.pop("display_kind", None)
api_msg.pop("display_kind", None)
api_msg.pop("display_metadata", None)
# Legacy hidden redirect placeholders (#88955): rows persisted
# BEFORE the writer-side api_content stamp in
# _apply_active_turn_redirect are content="" with no sidecar.
# Once display_kind is stripped the pre-call sanitizer
# (repair_empty_non_final_messages) would re-heal such a row on
# every call forever, since the durable transcript is never
# mutated. Give the wire copy the same neutral payload here so
# old sessions converge too. Never the interrupt scaffold —
# replaying scaffold bytes as assistant text is #81841.
if (
_display_kind == "hidden"
and api_msg.get("role") == "assistant"
and not _api_content
and not (api_msg.get("content") or "").strip()
and not api_msg.get("tool_calls")
):
from agent.agent_runtime_helpers import _INTERRUPTED_PLACEHOLDER
api_msg["content"] = _INTERRUPTED_PLACEHOLDER
# Durable row identity stamped by _rows_to_conversation so the
# desktop can address a specific persisted message (reactions).
# Bookkeeping, never a provider field — only the chat-completions
@@ -2411,6 +2536,16 @@ def run_conversation(
# Remove finish_reason - not accepted by strict APIs (e.g. Mistral)
if "finish_reason" in api_msg:
api_msg.pop("finish_reason")
# Empty non-final user/assistant turns (#88955 hidden placeholders
# and #96870 stream-death / host-fed empties): once display_kind
# and api_content are stripped, the pre-call sanitizer would
# re-heal the wire copy on every send and flood errors.log.
# Fill the WIRE copy here so the sanitizer has nothing to do.
# Durable history is not mutated. After reasoning copy so a
# thinking-only turn keeps its payload and is not rewritten.
fill_empty_non_final_wire_payload(
api_msg, is_final=(idx == len(messages) - 1)
)
# _thinking_prefill survives here intentionally: the drop pass below
# needs it. The transport strips all underscore keys before the wire.
# Strip length-continuation marks; not every transport drops underscore keys.
@@ -2556,6 +2691,24 @@ def run_conversation(
# manual message manipulation are always caught.
api_messages = agent._sanitize_api_messages(api_messages)
# One-time repeated-heal escalation notice (#96870): if the sanitizer
# above just crossed the per-session heal threshold, deliver the
# queued notice through the status/warning callback — the normal
# out-of-band delivery channel (gateway status message / CLI print).
# NEVER appended to messages/api_messages: conversation context and
# the cached prompt prefix stay byte-identical.
try:
from agent.agent_runtime_helpers import (
consume_pending_sanitizer_heal_notice,
)
_heal_notice = consume_pending_sanitizer_heal_notice()
if _heal_notice:
agent._emit_warning(_heal_notice)
except Exception:
# A notice hiccup must never break the send path.
logger.debug("sanitizer heal notice delivery failed", exc_info=True)
# Drop thinking-only assistant turns (reasoning but no visible
# output and no tool_calls) and merge any adjacent user messages
# left behind. Prevents Anthropic 400s ("The final block in an
@@ -2672,7 +2825,19 @@ def run_conversation(
# messages walk inside estimate_request_tokens_rough. Tools added
# separately (compression needs them: 50+ tools = 20-30K tokens).
# total_chars is a rough (~) proxy — verbose log + hook metric only.
approx_tokens = estimate_messages_tokens_rough(api_messages)
# Charge stale thinking only when the active route actually replays
# it (#84371): on codex_responses the text keys never ship (the
# encrypted item sidecars — charged unconditionally — carry the
# chain), so counting them here re-created the trigger/tail-walk
# disagreement that dead-looped compaction.
from agent.turn_context import _agent_stale_thinking_on_wire
if _agent_stale_thinking_on_wire(agent):
approx_tokens = estimate_messages_tokens_rough(api_messages)
else:
approx_tokens = estimate_messages_tokens_rough(
api_messages, charge_stale_thinking=False
)
# Route-aware pressure: when the upcoming request is eligible for
# native Responses compaction the transport will checkpoint-prune
# the payload before sending — the generic durable-history figure
@@ -2743,6 +2908,21 @@ def run_conversation(
_preflight_threshold = int(
getattr(_compressor, "threshold_tokens", 0) or 0
)
_provider_overflow_preflight = (
_provider_overflow_recovery_pending
and (
_preflight_threshold <= 0
or request_pressure_tokens >= _preflight_threshold
)
)
if (
_provider_overflow_recovery_pending
and not _provider_overflow_preflight
):
# The outer-loop rebuild includes the active system prompt,
# request-only injections, and tool schemas. Once that complete
# request has real output runway again, the provider may be tried.
_provider_overflow_recovery_pending = False
# A previous mid-turn preflight pass deliberately continued the loop so
# API-only context and all sanitization could be rebuilt. Compare that
# fully assembled request with the fully assembled request that caused
@@ -2782,11 +2962,50 @@ def run_conversation(
and not _review_fork_first_request_pending(agent)
and len(messages) > 1
and compression_attempts < max_compression_attempts
and not _preflight_compression_blocked
and not _defer_preflight(request_pressure_tokens)
and (
not _preflight_compression_blocked
or _provider_overflow_preflight
)
and (
not _defer_preflight(request_pressure_tokens)
or _provider_overflow_preflight
)
and not _compression_cooldown
and _compressor.should_compress(request_pressure_tokens)
):
# Managed local runtime: try GROWING the context window before
# compressing (the window ladder's design order — compression is
# the move of last resort, once the window is at the model's
# native max or physics/speed say stop). Only fires for a
# llamacpp-flavored provider whose base_url is the server this
# process supervises; every other provider falls straight
# through to compression, exactly as before.
_grown_window = _maybe_grow_local_window(
agent, _compressor, request_pressure_tokens
)
if _grown_window:
# The server now grants a bigger window: recalibrate the
# compressor to it and skip compression this pass — the
# request that was over the OLD threshold fits the new one.
_compressor.update_model(
agent.model,
_grown_window,
base_url=getattr(agent, "base_url", "") or "",
api_key=getattr(agent, "api_key", "") or "",
provider=getattr(agent, "provider", "") or "",
api_mode=getattr(agent, "api_mode", "") or "",
)
agent._buffer_status(
f"📈 Context window grown to {_grown_window // 1024}K "
f"(local model; conversation continues uncompressed)"
)
# This preflight iteration never reached the provider —
# refund the consumed call/budget exactly as the compression
# path below does before ITS continue.
api_call_count -= 1
agent._api_call_count = api_call_count
agent.iteration_budget.refund()
continue
if _moa_prepared_request is not None:
pending_moa_prepared_request = _moa_prepared_request
compression_attempts += 1
@@ -2836,6 +3055,22 @@ def run_conversation(
approx_tokens=request_pressure_tokens,
task_id=effective_task_id,
)
if context_compression_timed_out(agent):
# Host progress-aware timeout (#98722, salvaged from #98741):
# this preflight iteration never reached the provider. Refund
# its provisional call/budget exactly like a successful
# pre-API compaction, then stop before the unchanged oversized
# request reaches the provider — its overflow error would only
# invoke compression again on the same transcript with the
# wait budget already spent.
api_call_count -= 1
agent._api_call_count = api_call_count
agent.iteration_budget.refund()
final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE
failed = True
_compression_timeout_exhausted = True
_turn_exit_reason = "context_compression_timeout"
break
if messages is _pre_api_input and (
compression_skipped_due_to_lock(agent)
or compression_blocked_transiently(agent)
@@ -2901,6 +3136,34 @@ def run_conversation(
_turn_exit_reason = "compaction_handoff_not_actionable"
break
continue
elif _provider_overflow_preflight and _compression_cooldown:
# The provider already proved this request cannot fit, while the
# compressor is temporarily unavailable. Do not send the known-
# oversized request again; let the next user turn retry after the
# cooldown instead of turning this into compression exhaustion.
agent._persist_session(messages, conversation_history)
return _compression_deferred_result(
agent,
messages,
api_call_count,
reason="transient_block",
)
elif (
_provider_overflow_preflight
and compression_attempts >= max_compression_attempts
):
# Every bounded recovery pass has been consumed and the rebuilt
# request is still over threshold. Fail closed before another
# provider call; llama.cpp can silently truncate an oversized
# retry instead of returning a second actionable overflow error.
return _provider_overflow_exhausted_result(
agent,
messages,
conversation_history,
api_call_count,
request_pressure_tokens,
max_compression_attempts,
)
elif (
agent.compression_enabled
and len(messages) > 1
@@ -2955,6 +3218,20 @@ def run_conversation(
if callable(_warn_fn):
_warn_fn(request_pressure_tokens, _ctx_len)
if _provider_overflow_preflight:
# Any other gate that prevented the forced preflight (for example,
# an uncompressible one-message request) must also fail closed.
# Falling through would send a request that the provider already
# proved cannot fit.
return _provider_overflow_exhausted_result(
agent,
messages,
conversation_history,
api_call_count,
request_pressure_tokens,
max_compression_attempts,
)
# Thinking spinner for quiet mode (animated during API call)
thinking_spinner = None
@@ -3048,6 +3325,10 @@ def run_conversation(
try:
agent._reset_stream_delivery_tracking()
# Per-attempt first-chunk timestamp, refreshed each attempt so
# a stale value from a previous API call can never leak into
# the post_api_request hook (set again on stream success).
agent._last_api_first_chunk_at = None
# api_messages is built once, before this retry loop, while the
# primary provider is active. A mid-conversation fallback can
# switch to a require-side provider (DeepSeek / Kimi / MiMo) that
@@ -4310,6 +4591,20 @@ def run_conversation(
)
if _new_anchor is not None:
agent._usage_anchor = _new_anchor
# Turn-base anchor for display surfaces: the FIRST
# response of a turn carries minimal current-turn
# reasoning replay, so its prompt_tokens approximate
# the durable transcript cost (what the next turn
# inherits). Later same-turn responses inflate
# prompt_tokens with replayed thinking + tool
# scaffolding that evaporates at the turn boundary —
# anchoring the context meter here instead of on the
# last response removes the end-of-turn sawtooth
# (850K mid-loop -> 600K next turn) that users read
# as a broken compaction. Display-only: compression
# trigger math keeps using real last-request usage.
if api_call_count == 1:
agent._turn_base_usage_anchor = _new_anchor
_compression_threshold = int(
getattr(agent.context_compressor, "threshold_tokens", 0)
or 0
@@ -5530,6 +5825,12 @@ def run_conversation(
)
)
time.sleep(2)
# Same class as the generic overflow handler below:
# the provider proved the request does not fit the
# (now-reduced) window, and row count alone is not
# proof the rebuilt request does. Recheck the
# complete request before the next provider call.
_provider_overflow_recovery_pending = True
_retry.restart_with_compressed_messages = True
break
# Fall through to normal error handling if compression
@@ -6178,6 +6479,28 @@ def run_conversation(
agent, messages, api_call_count,
reason="transient_block",
)
if context_compression_timed_out(agent):
# Host progress-aware timeout (#98722, salvaged from
# #98741): the provider proved the request does not
# fit, but this recovery pass spent the full wait
# budget without a committed summary. Re-sending the
# unchanged request would bounce off the same overflow
# error and re-enter compression in the same turn. End
# the turn with the typed recovery contract instead —
# transcript intact, no further doomed provider sends.
agent._persist_session(messages, conversation_history)
_final_response = _COMPRESSION_TIMEOUT_FINAL_RESPONSE
return {
"final_response": _final_response,
"messages": messages,
"completed": False,
"api_calls": api_call_count,
"error": _final_response,
"partial": True,
"failed": True,
"compression_exhausted": True,
"turn_exit_reason": "context_compression_timeout",
}
conversation_history = conversation_history_after_compression(
agent, messages, conversation_history
)
@@ -6195,6 +6518,11 @@ def run_conversation(
elif new_tokens > 0 and new_tokens < original_tokens * 0.95:
agent._buffer_status(COMPRESSION_RETRY_TOKENS_STATUS_TEMPLATE.format(before=original_tokens, after=new_tokens))
time.sleep(2) # Brief pause between compression retries
# Rebuild the complete request before the next provider
# call and force normal preflight to honor it. Message
# count alone is not proof that system/tool-inclusive
# token pressure fell.
_provider_overflow_recovery_pending = True
_retry.restart_with_compressed_messages = True
break
else:
@@ -6993,6 +7321,14 @@ def run_conversation(
api_duration=api_duration,
started_at=api_start_time,
ended_at=_api_ended_at,
# First received stream chunk timestamp (epoch seconds), set by
# interruptible_streaming_api_call from its per-attempt
# stream diagnostics; None when the response was not
# streamed or no chunk arrived. TTFB =
# first_chunk_at - started_at.
first_chunk_at=getattr(
agent, "_last_api_first_chunk_at", None
),
finish_reason=finish_reason,
message_count=len(api_messages),
response_model=getattr(response, "model", None),
@@ -8795,7 +9131,7 @@ def run_conversation(
# Post-loop turn finalization extracted to agent/turn_finalizer.finalize_turn
# (god-file decomposition Phase 1 step 4). Behavior-neutral: the assembled
# result dict is returned exactly as before.
return finalize_turn(
result = finalize_turn(
agent,
final_response=final_response,
api_call_count=api_call_count,
@@ -8812,6 +9148,15 @@ def run_conversation(
_pending_verification_response=_pending_verification_response,
_pending_verification_response_previewed=_pending_verification_response_previewed,
)
if _compression_timeout_exhausted:
# Reuse the gateway's existing context-recovery contract (#98722,
# salvaged from #98741). The bloated transcript remains intact while
# future input can move to a clean session instead of replaying the
# summarize-timeout loop.
result["error"] = _COMPRESSION_TIMEOUT_FINAL_RESPONSE
result["partial"] = True
result["compression_exhausted"] = True
return result
+171 -42
View File
@@ -489,34 +489,83 @@ def _iter_custom_providers(config: Optional[dict] = None):
yield _normalize_custom_pool_name(name), entry
def get_custom_provider_pool_key(base_url: Optional[str], provider_name: Optional[str] = None) -> Optional[str]:
"""Look up the custom_providers list in config.yaml and return 'custom:<name>' for a matching base_url.
def _custom_entry_name_aliases(norm_name: str, entry: Dict[str, Any]) -> set:
aliases = {norm_name}
provider_key = _normalize_custom_pool_name(str(entry.get("provider_key") or ""))
if provider_key:
aliases.add(provider_key)
return aliases
When provider_name is given, prefer matching by name first (solving the case where
multiple custom providers share the same base_url but have different API keys).
Falls back to base_url matching when no name match is found.
Returns None if no match is found.
def _requested_custom_name_aliases(provider_name: str) -> set:
normalized = _normalize_custom_pool_name(provider_name)
aliases = {normalized} if normalized else set()
if normalized.startswith(CUSTOM_POOL_PREFIX):
suffix = _normalize_custom_pool_name(normalized[len(CUSTOM_POOL_PREFIX):])
if suffix:
aliases.add(suffix)
return aliases
def _pool_keys_for_custom_entry(norm_name: str, entry: Dict[str, Any]) -> List[str]:
"""Durable ``providers.<key>`` slug first, then legacy ``custom:<name>``."""
keys: List[str] = []
seen = set()
def _add(key: str) -> None:
normalized = str(key or "").strip().lower()
if normalized and normalized not in seen:
seen.add(normalized)
keys.append(normalized)
provider_key = _normalize_custom_pool_name(str(entry.get("provider_key") or ""))
if provider_key:
_add(provider_key)
if norm_name:
_add(f"{CUSTOM_POOL_PREFIX}{norm_name}")
return keys
def custom_provider_pool_key_candidates(
base_url: Optional[str],
provider_name: Optional[str] = None,
) -> List[str]:
"""Return pool keys to try for a custom endpoint.
``hermes auth add <key>`` stores new-style ``providers.<key>`` credentials
under the durable config slug (``b-ai``). Older rows and legacy
``custom_providers:`` entries still live under ``custom:<display-name>``.
Try the slug first, then the legacy namespace, so a populated pool is not
skipped in favour of the ``no-key-required`` placeholder.
"""
if not base_url:
return None
return []
normalized_url = base_url.strip().rstrip("/")
requested_aliases = (
_requested_custom_name_aliases(provider_name) if provider_name else set()
)
# When a provider name is given, try to match by name first.
# This fixes the P1 bug where two custom providers sharing the same
# base_url always resolve to the first one's credentials.
if provider_name:
normalized_name = _normalize_custom_pool_name(provider_name)
if requested_aliases:
for norm_name, entry in _iter_custom_providers():
if norm_name == normalized_name:
return f"{CUSTOM_POOL_PREFIX}{norm_name}"
if requested_aliases & _custom_entry_name_aliases(norm_name, entry):
return _pool_keys_for_custom_entry(norm_name, entry)
# Fall back to base_url matching (original behavior)
for norm_name, entry in _iter_custom_providers():
entry_url = str(entry.get("base_url") or "").strip().rstrip("/")
if entry_url and entry_url == normalized_url:
return f"{CUSTOM_POOL_PREFIX}{norm_name}"
return None
return _pool_keys_for_custom_entry(norm_name, entry)
return []
def get_custom_provider_pool_key(base_url: Optional[str], provider_name: Optional[str] = None) -> Optional[str]:
"""Look up the matching custom provider and return its preferred pool key.
Prefers the durable ``providers.<key>`` slug when present, otherwise
``custom:<normalized-name>``. When provider_name is given, match by name
first so two custom providers sharing a base_url keep separate keys.
"""
candidates = custom_provider_pool_key_candidates(base_url, provider_name)
return candidates[0] if candidates else None
def list_custom_pool_providers() -> List[str]:
@@ -557,6 +606,36 @@ def get_pool_strategy(provider: str) -> str:
return STRATEGY_FILL_FIRST
def _keyed_custom_pool_matches(
pool_provider: str,
provider_norm: str,
base_url: Optional[str],
) -> bool:
"""Match a durable ``providers.<key>`` pool against runtime identities."""
runtime_url = str(base_url or "").strip().rstrip("/")
if not runtime_url:
return False
try:
for normalized_name, entry in _iter_custom_providers():
provider_key = _normalize_custom_pool_name(
str(entry.get("provider_key") or "")
)
if provider_key != pool_provider:
continue
aliases = _custom_entry_name_aliases(normalized_name, entry)
aliases.add(f"{CUSTOM_POOL_PREFIX}{normalized_name}")
if provider_key:
aliases.add(f"{CUSTOM_POOL_PREFIX}{provider_key}")
configured_url = str(entry.get("base_url") or "").strip().rstrip("/")
if provider_norm == "custom":
return runtime_url == configured_url
runtime_aliases = _requested_custom_name_aliases(provider_norm)
return bool(runtime_aliases & aliases) and runtime_url == configured_url
except Exception:
return False
return False
def credential_pool_matches_provider(
pool_or_provider: Any,
provider: Optional[str],
@@ -567,10 +646,12 @@ def credential_pool_matches_provider(
Named custom endpoints may use three identities: the live agent can retain
the configured name/provider key, newer runtime paths normalize it to
``custom``, and the pool is keyed ``custom:<name>``. Accept those aliases
only when the runtime endpoint belongs to the same configured custom
provider. Empty identities fail closed. Legacy pool adapters without a
``provider`` attribute remain compatible; production pools are scoped.
``custom``, and the pool may be keyed either as the durable
``providers.<key>`` slug or as legacy ``custom:<name>``. Accept those
aliases only when the runtime endpoint belongs to the same configured
custom provider. Empty identities fail closed. Legacy pool adapters
without a ``provider`` attribute remain compatible; production pools
are scoped.
"""
raw_pool_provider = getattr(pool_or_provider, "provider", None)
if raw_pool_provider is None:
@@ -586,13 +667,18 @@ def credential_pool_matches_provider(
if not pool_provider or not provider_norm:
return False
if not pool_provider.startswith(CUSTOM_POOL_PREFIX):
return pool_provider == provider_norm
if pool_provider == provider_norm:
return True
return _keyed_custom_pool_matches(pool_provider, provider_norm, base_url)
if provider_norm == "custom":
try:
matched_pool = get_custom_provider_pool_key(base_url or "")
if str(matched_pool or "").strip().lower() == pool_provider:
return True
candidates = custom_provider_pool_key_candidates(base_url or "")
except Exception:
return False
return str(matched_pool or "").strip().lower() == pool_provider
return pool_provider in {str(key).strip().lower() for key in candidates}
runtime_url = str(base_url or "").strip().rstrip("/")
if not runtime_url:
@@ -625,9 +711,10 @@ def credential_pool_matches_provider(
def resolve_runtime_pool_key(provider: Optional[str], base_url: Optional[str]) -> str:
"""Resolve the credential-pool key for a runtime provider identity.
Named custom runtimes retain their configured alias while their pool is
stored under ``custom:<name>``. Return that scoped key only when the
canonical provider/endpoint boundary accepts it; otherwise preserve the
Named custom runtimes retain their configured alias while their pool may
be stored under the durable ``providers.<key>`` slug or legacy
``custom:<name>``. Return that scoped key only when the canonical
provider/endpoint boundary accepts it; otherwise preserve the
normalized runtime identity so callers fail closed.
"""
provider_norm = str(provider or "").strip().lower()
@@ -644,18 +731,19 @@ def resolve_runtime_pool_key(provider: Optional[str], base_url: Optional[str]) -
):
return str(candidate).strip().lower()
else:
# Named and exact custom runtimes are keyed by provider identity,
# while auth storage remains keyed by display name. Search the
# configured candidates by identity before considering endpoint;
# this prevents a sibling sharing the URL from lending its pool.
for normalized_name, _entry in _iter_custom_providers():
candidate = f"{CUSTOM_POOL_PREFIX}{normalized_name}"
if credential_pool_matches_provider(
candidate,
provider_norm,
base_url=base_url,
):
return candidate
# Named and exact custom runtimes are keyed by provider identity.
# Auth storage prefers the durable providers.<key> slug, with
# legacy custom:<display-name> as fallback. Search configured
# candidates by identity before considering endpoint so a sibling
# sharing the URL cannot lend its pool.
for normalized_name, entry in _iter_custom_providers():
for candidate in _pool_keys_for_custom_entry(normalized_name, entry):
if credential_pool_matches_provider(
candidate,
provider_norm,
base_url=base_url,
):
return candidate
except Exception:
pass
return provider_norm
@@ -3265,6 +3353,35 @@ def get_env_prefer_dotenv(key: str) -> str:
return raw or scoped_value
# Providers we've already warned about env-key → pool ingestion for, once per
# process. See _warn_env_ingestion_once (#81952 expected-behavior #3).
_ENV_INGESTION_WARNED: Set[str] = set()
def _warn_env_ingestion_once(provider: str, env_var: str) -> None:
"""WARN (once per process per provider) when an env credential is newly
ingested into a paid provider's pool.
Auto-ingesting OPENROUTER_API_KEY is what ARMS silent OpenRouter spend —
every downstream auto-detect (resolve_provider pool probe, aux fallback)
keys off the pool having credentials. Ingestion itself stays allowed (the
user exported the key = arguable intent, per #81952), but it must never be
silent.
"""
if provider in _ENV_INGESTION_WARNED:
return
_ENV_INGESTION_WARNED.add(provider)
logger.warning(
"Ingested %s from environment into the %s credential pool — this "
"enables %s spend. Remove the key or run "
"hermes auth remove %s <n> to suppress.",
env_var,
provider,
"OpenRouter" if provider == "openrouter" else provider,
provider,
)
def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool, Set[str]]:
changed = False
active_sources: Set[str] = set()
@@ -3335,7 +3452,7 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool
if _is_source_suppressed(provider, source):
return changed, active_sources
active_sources.add(source)
changed |= _upsert_entry(
ingested = _upsert_entry(
entries,
provider,
source,
@@ -3346,6 +3463,9 @@ def _seed_from_env(provider: str, entries: List[PooledCredential]) -> Tuple[bool
base_url=OPENROUTER_BASE_URL,
),
)
changed |= ingested
if ingested:
_warn_env_ingestion_once(provider, "OPENROUTER_API_KEY")
return changed, active_sources
pconfig = PROVIDER_REGISTRY.get(provider)
@@ -3476,9 +3596,18 @@ def _seed_custom_pool(pool_key: str, entries: List[PooledCredential]) -> Tuple[b
model_api_key = v.strip()
break
if model_provider == "custom" and model_base_url and model_api_key:
# Check if this model's base_url matches our custom provider
matched_key = get_custom_provider_pool_key(model_base_url)
if matched_key == pool_key:
# Check if this model's base_url matches our custom provider.
# The pool may be keyed under either the durable
# ``providers.<key>`` slug or the legacy ``custom:<name>``
# namespace, so accept the match against any candidate —
# comparing against the single preferred key silently skips
# seeding when the pool holds the other identity (verified
# regression from PR #100413 review).
matched_keys = {
str(key).strip().lower()
for key in custom_provider_pool_key_candidates(model_base_url)
}
if pool_key in matched_keys:
source = "model_config"
if not _is_suppressed(pool_key, source):
active_sources.add(source)
+3 -8
View File
@@ -252,15 +252,10 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool:
if not base_url:
return False
try:
from hermes_cli.models import _is_model_free, _pricing_cache
from hermes_cli.models import _is_model_free, peek_cached_pricing
# Mirror get_pricing_for_provider's key normalization: the agent's
# Nous base_url is /v1-suffixed (https://inference-api.nousresearch.com/v1)
# but the picker keys _pricing_cache on the pre-/v1 root.
key = base_url.rstrip("/")
if key.endswith("/v1"):
key = key[:-3].rstrip("/")
pricing = _pricing_cache.get(key)
# peek_cached_pricing owns the /v1-suffix and auth-state key details.
pricing = peek_cached_pricing(base_url)
if not pricing:
return False
return _is_model_free(model, pricing)
+27 -11
View File
@@ -411,9 +411,6 @@ CURATOR_DRY_RUN_BANNER = (
"\n"
" • DO NOT call skill_manage with action=patch, create, delete, "
"write_file, or remove_file.\n"
" • DO NOT call terminal to mv skill directories into .archive/.\n"
" • DO NOT call terminal to mv, cp, rm, or rewrite any file under "
"~/.hermes/skills/.\n"
" • skills_list and skill_view are FINE — read as much as you need.\n"
"\n"
"Your output IS the deliverable. Produce the exact same "
@@ -508,9 +505,14 @@ CURATOR_REVIEW_PROMPT = (
"copied and modified\n"
" • `scripts/<name>.<ext>` for statically re-runnable actions "
"(verification scripts, fixture generators, probes)\n"
" Then archive the old sibling. Use `terminal` with `mkdir -p "
"~/.hermes/skills/<umbrella>/references/ && mv ... <umbrella>/"
"references/<topic>.md` (or templates/ / scripts/).\n\n"
" Then archive the old sibling. Re-home the content through the "
"LEDGERED tool surface: `skill_manage action=write_file` on the umbrella "
"to place the file (subdirectories are created for you), then "
"`skill_manage action=remove_file` on the source to drop the original, "
"then `skill_manage action=delete` on the source. Never a terminal move "
"— a shell mv/cp writes the same bytes with no ledger entry, so the "
"archive that follows snapshots an already-stripped package and "
"`hermes curator rollback` restores a hollow skill (issue #96962).\n\n"
"Package integrity — not optional:\n"
"Before demoting or archiving a skill, inspect it as a COMPLETE "
"directory package, not just SKILL.md. A skill root may include "
@@ -556,9 +558,10 @@ CURATOR_REVIEW_PROMPT = (
"skill, or `absorbed_into=\"\"` when you're truly pruning with no "
"forwarding target. This drives cron-job skill-reference migration — "
"guessing from your YAML summary after the fact is fragile.\n"
" - terminal — move LOCAL candidate content into "
"a support subfile when package integrity requires it; never mv, cp, rm, "
"patch, or rewrite bundled, hub-installed, or external-dir skills\n\n"
" You have NO terminal access in this pass — every filesystem mutation "
"goes through skill_manage above so it is ledgered and rollback-able "
"(issue #96962). Reading files works through skill_view (including "
"skill_view(name, file_path=...) for support files).\n\n"
"'keep' is a legitimate decision ONLY when the skill is already a "
"class-level umbrella and none of the proposed merges would improve "
"discoverability. 'This is narrow but distinct from its siblings' "
@@ -1543,7 +1546,9 @@ def run_curator_review(
If *dry_run* is True, the automatic stale/archive transitions are SKIPPED
and the LLM review pass is instructed to produce a report only — no
skill_manage mutations, no terminal archive moves. The REPORT.md still
skill_manage mutations. (The fork has no terminal access at all — see the
``enabled_toolsets=["skills"]`` kwarg in ``_run_llm_review``.) The
REPORT.md still
gets written and ``state.last_report_path`` still records it so users
can read what the curator WOULD have done. A dry-run also honors
*consolidate*: when consolidation is off, the preview only reports the
@@ -1945,7 +1950,18 @@ def _run_llm_review(prompt: str) -> Dict[str, Any]:
credential_pool=_credential_pool,
request_overrides=_request_overrides,
**_agent_kwargs,
enabled_toolsets=["skills", "terminal"],
enabled_toolsets=["skills"],
# ``terminal`` was deliberately removed from this fork (issue
# #96962): a terminal ``mv``/``cp``/``rm`` under the skills tree
# writes the same bytes with NO ledger entry, so the archive that
# followed snapshotted an already-stripped package and ``hermes
# curator rollback`` restored a hollow skill. Every mutation this
# fork needs has a ledgered skill_manage action (write_file /
# remove_file / delete), and reading works through skill_view.
# Removing the toolset closes the hole by construction — no
# command-parsing heuristic to evade, no process stdin to feed,
# no remote-backend divergence — which no terminal-write guard
# over a Turing-complete input space can guarantee.
# Umbrella-building over a large skill collection is worth a
# high iteration ceiling — the pass typically takes 50-100
# API calls against hundreds of candidate skills. The
+25 -2
View File
@@ -63,6 +63,7 @@ Design invariants:
from __future__ import annotations
import asyncio
import contextvars
import faulthandler
import logging
import os
@@ -100,6 +101,12 @@ MAX_SAFE_TIMEOUT_S = 31_536_000.0 # 365 days
# is blocked in a synchronous call and dumping stacks (family A diagnostics).
_LOOP_BLOCKED_DUMP_GRACE_S = 5.0
# ``Event.wait`` is a C-level block: KeyboardInterrupt / SetAsyncExc only
# land when the thread returns to Python. Slice the wait so a /stop or
# SIGINT during a bounded sync call is observed within this window rather
# than at the full deadline (#94285, tools/test_local_interrupt_cleanup).
_BOUNDED_SYNC_WAIT_SLICE_S = 0.2
class DeadlineExpired(TimeoutError):
"""A deadline enforced by this layer expired.
@@ -485,6 +492,14 @@ def run_bounded_sync(
in a retry loop would accumulate them.
``timeout=None`` (or non-positive) blocks until ``fn`` returns.
The worker runs under ``contextvars.copy_context()`` so profile secret
scope, session id, and delegated-child guards set on the caller survive
the thread hop (terminal env.execute — #94285 CI).
The caller's wait is sliced (``_BOUNDED_SYNC_WAIT_SLICE_S``) so a
``KeyboardInterrupt`` or ``PyThreadState_SetAsyncExc`` lands within
that window instead of only at the full deadline.
"""
timeout_s = clamp_timeout(timeout)
start = time.monotonic()
@@ -499,10 +514,11 @@ def run_bounded_sync(
box: dict[str, Any] = {}
done = threading.Event()
ctx = contextvars.copy_context()
def _worker() -> None:
try:
box["value"] = fn()
box["value"] = ctx.run(fn)
except BaseException as exc: # re-raised in caller; must not vanish
box["exc"] = exc
finally:
@@ -510,7 +526,14 @@ def run_bounded_sync(
thread = threading.Thread(target=_worker, name=f"deadline-{label}", daemon=True)
thread.start()
if not done.wait(timeout_s):
deadline = start + timeout_s
while not done.is_set():
remaining = deadline - time.monotonic()
if remaining <= 0:
break
done.wait(min(_BOUNDED_SYNC_WAIT_SLICE_S, remaining))
if not done.is_set():
logger.warning(
"[deadline] %r timed out after %.1fs; worker abandoned", label, timeout_s
)
+90 -33
View File
@@ -10,9 +10,9 @@
agent run (``gateway/run.py:_handle_message``).
In-flight work is NEVER killed — this is pause-new-work, not panic/exit.
The check is a single ``os.stat`` so callers may run it every tick; no
caching beyond the OS is performed, so engaging/disengaging takes effect on
the very next check.
The check is one or two ``os.stat`` calls (process home + fleet root when
they differ) so callers may run it every tick; no caching beyond the OS is
performed, so engaging/disengaging takes effect on the very next check.
The sentinel body is optional JSON ``{"reason": ..., "engaged_at": ...}``.
A corrupt or empty file still counts as engaged (fail safe): the pause must
@@ -51,24 +51,60 @@ def _hermes_home() -> Path:
return Path(os.path.expanduser("~/.hermes"))
def _canonical_root() -> Path:
"""Fleet-wide Hermes root, even when this process is a profile gateway.
Profile gateways launch with HERMES_HOME=~/.hermes/profiles/<name>.
``hermes pause`` from an operator seat writes ~/.hermes/ESTOP. If we
only inspect the profile home, the emergency stop does not bind
(jarvis-os/t_7b65ff88: fleet-analyst kept dispatching through pause).
"""
try:
from hermes_constants import get_default_hermes_root
return Path(get_default_hermes_root())
except Exception:
return Path(os.path.expanduser("~/.hermes"))
def sentinel_path() -> Path:
"""Path of the ESTOP sentinel under the active HERMES_HOME."""
"""Path of the ESTOP sentinel this process would write on `hermes pause`."""
return _hermes_home() / SENTINEL_NAME
def is_engaged() -> bool:
"""Cheap check (one stat): is the global emergency stop engaged?
Fail SAFE on stat errors: if we cannot determine whether the sentinel
exists (permission error, transient I/O failure on HERMES_HOME), report
engaged. The module contract is that the pause must hold even when the
sentinel is unreadable — a fail-open here would silently lift an
operator's emergency stop exactly when the filesystem is misbehaving.
"""
def _candidate_sentinel_paths() -> list:
"""Profile home first, then the fleet root if it is a different directory."""
primary = sentinel_path()
paths = [primary]
try:
return sentinel_path().exists()
except OSError:
return True
root = _canonical_root() / SENTINEL_NAME
except Exception:
return paths
try:
if root.resolve() != primary.resolve():
paths.append(root)
except Exception:
# Non-Path test doubles (fail-safe stat fixture) fail .resolve();
# the generic comparison below still dedupes plain equal paths.
if root != primary:
paths.append(root)
return paths
def is_engaged() -> bool:
"""Cheap check: is the global emergency stop engaged?
Engaged if ANY candidate sentinel exists: the process HERMES_HOME
(profile-local) or the fleet canonical root (~/.hermes). Fail SAFE on
stat errors so an unreadable sentinel still holds the pause.
"""
saw_stat_error = False
for path in _candidate_sentinel_paths():
try:
if path.exists():
return True
except OSError:
saw_stat_error = True
return saw_stat_error
def engage(reason: Optional[str] = None) -> Path:
@@ -91,14 +127,22 @@ def engage(reason: Optional[str] = None) -> Path:
def disengage() -> bool:
"""Remove the ESTOP sentinel. Returns True if a pause was lifted."""
try:
sentinel_path().unlink()
return True
except FileNotFoundError:
return False
except OSError:
return False
"""Remove ESTOP sentinels this process can see.
Lifts both the process-local sentinel and the fleet-root sentinel so
``hermes resume`` from a profile gateway still clears an operator pause
written at ~/.hermes/ESTOP.
"""
lifted = False
for path in _candidate_sentinel_paths():
try:
path.unlink()
lifted = True
except FileNotFoundError:
continue
except (OSError, AttributeError):
continue
return lifted
def get_state() -> Optional[dict]:
@@ -107,18 +151,31 @@ def get_state() -> Optional[dict]:
A sentinel with an unreadable/corrupt body still reports engaged, with
both fields None — the pause is authoritative, the metadata is not.
"""
path = sentinel_path()
if not path.exists():
if not is_engaged():
return None
reason = None
engaged_at = None
try:
raw = json.loads(path.read_text(encoding="utf-8"))
if isinstance(raw, dict):
reason = raw.get("reason") or None
engaged_at = raw.get("engaged_at") or None
except (OSError, ValueError):
pass
found = False
for path in _candidate_sentinel_paths():
try:
exists = path.exists()
except OSError:
return {"reason": None, "engaged_at": None}
except AttributeError:
continue
if not exists:
continue
found = True
try:
raw = json.loads(path.read_text(encoding="utf-8"))
if isinstance(raw, dict):
reason = raw.get("reason") or None
engaged_at = raw.get("engaged_at") or None
break
except (OSError, ValueError, AttributeError):
continue
if not found:
return None
return {"reason": reason, "engaged_at": engaged_at}
+44 -3
View File
@@ -519,6 +519,28 @@ def _lookup_supports_vision(
return override
if not provider or not model:
return None
# Managed local runtime: the server that would receive the image is
# the authority on whether it can see (its /props reports modalities
# when a vision projector is loaded; the catalog covers staged-but-
# unloaded models). Cloud catalogs have never heard of a local GGUF,
# so without this answer every local model reads as text-only and
# images detour to a cloud auxiliary — wrong twice for a local-first
# user (broken feature, and a screenshot leaving the machine).
try:
from hermes_cli.local_runtime.capabilities import (
is_managed_provider,
managed_model_supports_vision,
)
if is_managed_provider(provider, _resolve_inference_base_url(cfg, provider) or ""):
managed = managed_model_supports_vision(model)
if managed is not None:
return managed
except Exception as exc: # pragma: no cover - defensive
logger.debug("image_routing: managed-runtime caps lookup failed for %s:%s — %s",
provider, model, exc)
caps = None
try:
from agent.models_dev import get_model_capabilities
@@ -813,12 +835,31 @@ def _file_to_data_url(path: Path) -> Optional[str]:
logger.warning("image_routing: failed to read %s — %s", path, exc)
return None
mime = _guess_mime(path, raw=raw)
if mime not in _UNIVERSALLY_SUPPORTED_MIMES:
accepted = _UNIVERSALLY_SUPPORTED_MIMES
# The managed local server decodes fewer formats than cloud providers
# (no WebP — and a WebP part fails SILENTLY: the model never sees an
# image and confabulates a description). When the active main model is
# served by the managed runtime, narrow the accepted set so those
# formats transcode to PNG here instead of vanishing server-side.
try:
from agent.auxiliary_client import _runtime_main_value
from hermes_cli.local_runtime.capabilities import (
ACCEPTED_IMAGE_MIMES,
is_managed_provider,
)
if is_managed_provider(
str(_runtime_main_value("provider") or ""),
str(_runtime_main_value("base_url") or "")):
accepted = ACCEPTED_IMAGE_MIMES
except Exception: # noqa: BLE001 — best-effort narrowing only
pass
if mime not in accepted:
transcoded = _transcode_to_png(raw)
if transcoded is None:
logger.warning(
"image_routing: %s is %s which is not accepted by all major "
"vision providers and could not be transcoded to PNG; "
"image_routing: %s is %s which is not accepted by the "
"active provider and could not be transcoded to PNG; "
"skipping this attachment.",
path, mime,
)
+44 -1
View File
@@ -41,6 +41,36 @@ from tools.registry import tool_error
# historical best-effort contract (API v1).
_LEGACY_PRE_COMPRESS_API_VERSION = 1
def _accepts_require_checkpoint(fn: Callable[..., Any]) -> bool:
"""True if ``fn`` can receive the ``require_checkpoint`` keyword.
Checkpoint (v2) providers written against the original docs example use
the bare ``on_pre_compress(self, messages)`` signature; calling them with
the keyword would raise ``TypeError`` — which, under
``require_checkpoint=True``, the host would re-raise as a checkpoint
failure even though the provider's durable write succeeded. Inspect the
signature and fall back to the legacy call shape when the keyword (or a
``**kwargs`` catch-all) is absent. Unreadable signatures (C callables,
exotic proxies) conservatively report False.
"""
try:
sig = inspect.signature(fn)
except (TypeError, ValueError):
return False
for param in sig.parameters.values():
if param.kind is inspect.Parameter.VAR_KEYWORD:
return True
if (
param.name == "require_checkpoint"
and param.kind in (
inspect.Parameter.KEYWORD_ONLY,
inspect.Parameter.POSITIONAL_OR_KEYWORD,
)
):
return True
return False
logger = logging.getLogger(__name__)
# How long shutdown_all() waits for in-flight background sync/prefetch work
@@ -1122,7 +1152,20 @@ class MemoryManager:
if is_checkpoint_provider and evidence_messages is not None:
provider_messages = evidence_messages
try:
result = provider.on_pre_compress(provider_messages)
if is_checkpoint_provider and _accepts_require_checkpoint(
provider.on_pre_compress
):
result = provider.on_pre_compress(
provider_messages,
require_checkpoint=require_checkpoint,
)
else:
# Legacy (v1) providers keep the strict one-argument
# contract. v2 providers written against the original
# docs example (``def on_pre_compress(self, messages)``)
# also land here instead of dying on an unexpected
# kwarg — they simply never see the requirement signal.
result = provider.on_pre_compress(provider_messages)
if result and result.strip():
parts.append(result)
except Exception as e:
+29
View File
@@ -624,6 +624,7 @@ __all__ = [
"reasoning_echo_family",
"matches_reasoning_echo_family",
"needs_reasoning_echo",
"stale_thinking_reaches_wire",
"apply_reasoning_content_policy",
"reapply_reasoning_echo",
]
@@ -893,6 +894,34 @@ def needs_reasoning_echo(provider: Any, model: Any, base_url: Any) -> bool:
return reasoning_echo_family(provider, model, base_url) is not None
def stale_thinking_reaches_wire(
api_mode: Any, provider: Any, model: Any, base_url: Any
) -> bool:
"""True when stale assistant ``reasoning``/``reasoning_content`` text is
actually replayed on the wire for the active route.
This is the single wire-truth predicate the compaction TRIGGER estimator
and the tail-budget walks must share (#84371): when they disagree, a
reasoning-heavy session can simultaneously look over-threshold to
preflight and fully tail-protected to the walk — an infinite ineffective
compaction loop.
* ``codex_responses``: the Responses input builder
(``_chat_messages_to_responses_input``) never reads the text keys —
reasoning continuity rides the encrypted ``codex_reasoning_items``
sidecar, which both estimators already charge unconditionally. Stale
thinking TEXT never ships → ``False``.
* chat-completions echo-back families (DeepSeek/Kimi/MiMo thinking
mode): ``apply_reasoning_content_policy`` replays the stored
``reasoning_content`` verbatim on EVERY assistant turn → ``True``.
* everything else: stripped or one-space-padded at send time (#73624)
→ ``False``.
"""
if (api_mode or "") == "codex_responses":
return False
return needs_reasoning_echo(provider, model, base_url)
def apply_reasoning_content_policy(
source_msg: dict, api_msg: dict, needs_thinking_pad: bool
) -> None:
+152 -3
View File
@@ -1502,6 +1502,35 @@ def fetch_endpoint_model_metadata(
model_alias = props.get("model_alias", "")
if n_ctx and model_alias and model_alias in cache:
cache[model_alias]["context_length"] = n_ctx
else:
# Router mode: bare /props 400s and telemetry is
# per-child (?model=). Enumerate children via the
# native /models (carries status) and read each
# LOADED child's granted window — the value the
# context policy actually granted, which the meter
# and compressor must follow. Unloaded children are
# skipped: probing them could trigger an autoload.
native = requests.get(base + "/models", headers=headers, timeout=5, verify=_verify)
if native.ok:
children = (native.json() or {}).get("data", [])
for child in children[:16]:
if not isinstance(child, dict):
continue
child_id = child.get("id")
status = (child.get("status") or {}).get("value")
if not child_id or child_id not in cache or status not in ("loaded", "ready"):
continue
pr = requests.get(
base + "/v1/props", params={"model": child_id},
headers=headers, timeout=5, verify=_verify)
if not pr.ok:
pr = requests.get(
base + "/props", params={"model": child_id},
headers=headers, timeout=5, verify=_verify)
if pr.ok:
child_ctx = (pr.json().get("default_generation_settings") or {}).get("n_ctx")
if child_ctx:
cache[child_id]["context_length"] = child_ctx
except Exception:
pass
@@ -1679,6 +1708,7 @@ def parse_context_limit_from_error(error_msg: str) -> Optional[int]:
- "context_length_exceeded: 131072"
- "Maximum context size 32768 exceeded"
- "model's max context length is 65536"
- "input token count is 32825 but model only supports up to 32768"
"""
error_lower = error_msg.lower()
# Pattern: look for numbers near context-related keywords
@@ -1690,6 +1720,12 @@ def parse_context_limit_from_error(error_msg: str) -> Optional[int]:
r'(\d{4,})\s*(?:token)?\s*(?:context|limit)',
r'>\s*(\d{4,})\s*(?:max|limit|token)', # "250000 tokens > 200000 maximum"
r'(\d{4,})\s*(?:max(?:imum)?)\b', # "200000 maximum"
# Google Gemini/Gemma: "Unable to submit request because the input
# token count is 32825 but model only supports up to 32768." The
# limit is the number AFTER "supports up to" — the input count that
# precedes it must not be captured, so this pattern anchors on the
# "supports up to" phrase itself.
r'supports?\s+(?:only\s+)?up\s+to\s+(\d{4,})',
]
for pattern in patterns:
match = re.search(pattern, error_lower)
@@ -2368,6 +2404,27 @@ def _query_local_context_length_uncached(model: str, base_url: str, api_key: str
return int(ctx)
break
# llama.cpp: /props reports default_generation_settings.n_ctx —
# the RUNTIME window the server grants. Critically, the router
# answers this (from its preset) even for a model that is not
# currently loaded, while /v1/models reports meta=null until
# load. Without this probe, resolving a lazily-loaded model at
# session start finds no metadata and falls through to the
# name-pattern defaults, where a family catch-all (e.g. "qwen"
# = 131072) misreports a server launched at 262144.
if server_type == "llamacpp":
for props_path in (f"/props?model={model}", "/props"):
try:
resp = client.get(f"{server_url}{props_path}")
except httpx.HTTPError:
break
if resp.status_code != 200:
continue
n_ctx = (resp.json().get("default_generation_settings")
or {}).get("n_ctx")
if isinstance(n_ctx, (int, float)) and n_ctx:
return int(n_ctx)
# LM Studio / vLLM / llama.cpp / Anthropic-compat proxies:
# try /v1/models/{model}
resp = client.get(f"{server_url}/v1/models/{model}")
@@ -3544,7 +3601,9 @@ def estimate_tokens_rough(text: str) -> int:
return dense + ((sparse + 3) // 4)
def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
def estimate_messages_tokens_rough(
messages: List[Dict[str, Any]], *, charge_stale_thinking: bool = True,
) -> int:
"""Rough token estimate for a message list (pre-flight only).
Image parts (base64 PNG/JPEG) are counted as a flat ~1500 tokens per
@@ -3552,6 +3611,19 @@ def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
character length. Without this, a single ~1MB screenshot would be
estimated at ~250K tokens and trigger premature context compression.
``charge_stale_thinking`` mirrors the tail-budget walk's policy
(``context_compressor._estimate_msg_budget_tokens``, #73624): generic
thinking text (``reasoning`` / ``reasoning_content``) rides the wire for
at most the NEWEST assistant turn on routes that do not echo stale
reasoning back (Codex Responses ships encrypted ``codex_reasoning_items``
instead of the text keys; strict chat-completions providers strip or
one-space-pad the field). Passing ``False`` excludes those keys on every
assistant turn but the newest, so the compaction TRIGGER sees the same
size class as the tail-protection walk — the disagreement made
reasoning-heavy codex_responses sessions fire preflight forever while the
walk found nothing to compact (#84371 dead loop). Default ``True``
preserves the conservative full charge for callers without route context.
Per-message results are memoized (see ``_estimate_message_tokens_cached``)
keyed on a deep *identity fingerprint* of the message, so re-walking a
long history every iteration only pays for messages whose object graph
@@ -3559,12 +3631,50 @@ def estimate_messages_tokens_rough(messages: List[Dict[str, Any]]) -> int:
leaf objects and structure, hence an identical estimate.
"""
_IMAGE_TOKEN_COST = 1500
if not charge_stale_thinking:
messages = _strip_stale_thinking_for_estimate(messages)
total = 0
for msg in messages:
total += _estimate_message_tokens_cached(msg, _IMAGE_TOKEN_COST)
return total
# Generic thinking-text keys replayed for at most the newest assistant turn
# on non-echo routes — must stay in lockstep with
# ``context_compressor._NEWEST_TURN_ONLY_BUDGET_KEYS``.
_STALE_THINKING_ESTIMATE_KEYS = ("reasoning", "reasoning_content")
def _strip_stale_thinking_for_estimate(
messages: List[Dict[str, Any]],
) -> List[Dict[str, Any]]:
"""Copy of ``messages`` with stale thinking keys removed (newest kept).
Shallow stripped copies share the original value objects, so the
per-message memo still hits for the stripped shape on subsequent walks.
"""
newest = -1
for i in range(len(messages) - 1, -1, -1):
m = messages[i]
if isinstance(m, dict) and m.get("role") == "assistant":
newest = i
break
out: List[Dict[str, Any]] = []
for i, m in enumerate(messages):
if (
i != newest
and isinstance(m, dict)
and m.get("role") == "assistant"
and any(m.get(k) for k in _STALE_THINKING_ESTIMATE_KEYS)
):
m = {
k: v for k, v in m.items()
if k not in _STALE_THINKING_ESTIMATE_KEYS
}
out.append(m)
return out
# --- Per-message token-estimate memo -------------------------------------
#
# ``estimate_messages_tokens_rough`` is called on the full history every
@@ -3692,10 +3802,24 @@ def _wire_message_shadow(msg: Dict[str, Any]) -> Dict[str, Any]:
and bool(sidecar)
and msg.get("role") in ("user", "assistant")
)
# The internal ``reasoning`` key never ships: every request build pops it
# after (optionally) promoting it into ``reasoning_content`` (see
# ``apply_reasoning_content_policy`` / conversation_loop's api_messages
# build). When a message carries BOTH keys — the normal shape on
# reasoning-echo providers, which pin ``reasoning_content`` at creation
# time while ``reasoning`` holds the same text for trajectory storage —
# counting both charged the same thinking twice and inflated the rough
# estimate by up to +53% against provider-reported prompt_tokens
# (#84371 comment data, llama.cpp/Qwen). Keep ``reasoning`` only as the
# promotion proxy when no ``reasoning_content`` exists to displace it.
_rc = msg.get("reasoning_content")
drop_reasoning_dup = isinstance(_rc, str) and bool(_rc.strip())
shadow: Dict[str, Any] = {}
for k, v in msg.items():
if k in ("_anthropic_content_blocks", "reasoning_details") or k in PERSISTENCE_ONLY_MESSAGE_FIELDS:
continue
if k == "reasoning" and drop_reasoning_dup:
continue
if k == "api_content":
# Always popped before the request is built; only counted when it
# actually replaces ``content``.
@@ -3750,6 +3874,7 @@ def estimate_request_tokens_rough(
*,
system_prompt: str = "",
tools: Optional[List[Dict[str, Any]]] = None,
charge_stale_thinking: bool = True,
) -> int:
"""Rough token estimate for a full chat-completions request.
@@ -3758,12 +3883,25 @@ def estimate_request_tokens_rough(
tools enabled, schemas alone can add 20-30K tokens — a significant
blind spot when only counting messages. Image content is counted
at a flat per-image cost (see estimate_messages_tokens_rough).
``charge_stale_thinking`` is forwarded to
``estimate_messages_tokens_rough`` — pass ``False`` when the active
route provably strips stale assistant thinking at send time (see
``message_sanitization.stale_thinking_reaches_wire``, #84371).
"""
total = 0
if system_prompt:
total += estimate_tokens_rough(system_prompt)
if messages:
total += estimate_messages_tokens_rough(messages)
if charge_stale_thinking:
# Positional-compatible call: test seams and plugin engines
# monkeypatch estimate_messages_tokens_rough with (messages)-only
# signatures; only the route-aware False path needs the kwarg.
total += estimate_messages_tokens_rough(messages)
else:
total += estimate_messages_tokens_rough(
messages, charge_stale_thinking=False
)
if tools:
total += _estimate_tools_tokens_rough(tools)
return total
@@ -3820,6 +3958,8 @@ def capture_usage_anchor(
def anchored_context_tokens(
messages: List[Dict[str, Any]],
anchor: Optional[Dict[str, Any]],
*,
charge_stale_thinking: bool = True,
) -> Optional[int]:
"""Context size anchored on the last provider-reported usage.
@@ -3829,6 +3969,13 @@ def anchored_context_tokens(
estimation). The assistant reply produced by the anchored response
(first appended message after the base) is skipped: its cost is already
counted exactly by ``completion_tokens``.
``charge_stale_thinking`` is forwarded to the delta estimate — pass
``False`` to exclude transient ``reasoning``/``reasoning_content`` text
on all but the newest assistant message in the delta (the durable-
transcript view used by display surfaces; see the turn-base anchor in
``agent/conversation_loop.py``). Default ``True`` preserves the
conservative full charge for request-size callers.
"""
if not isinstance(anchor, dict) or not isinstance(messages, list):
return None
@@ -3850,7 +3997,9 @@ def anchored_context_tokens(
# completion_tokens above.
delta = delta[1:]
if delta:
total += estimate_messages_tokens_rough(delta)
total += estimate_messages_tokens_rough(
delta, charge_stale_thinking=charge_stale_thinking
)
return total
+42
View File
@@ -56,6 +56,48 @@ _conversation_id: ContextVar[Optional[str]] = ContextVar(
)
# ── Ambient routing/affinity scope ───────────────────────────────────────────
#
# Separate from the conversation id above, which is an ATTRIBUTION value: it
# names the conversation a request belongs to and is sent to the Portal as
# ``conversation=<id>``. The affinity scope is a ROUTING value — OpenRouter's
# sticky ``session_id``, Nous Portal's sticky key and xAI's ``x-grok-conv-id``
# use it to pin one conversation to one backend/prompt cache.
#
# The two agree for every host that keeps one session id per conversation, so
# the providers historically read the attribution id for both. They diverge
# for a host that mints one physical session per RESPONSE: attribution still
# resolves per row, while routing must follow the key the host declared for
# the whole chat (``agent.prompt_cache_scope.declared_conversation_scope``,
# issue #96811). Only that declared value is published here — unset means
# "no declaration", and consumers fall back to the conversation id exactly as
# before, so delegate trees keep sharing their parent's sticky key.
_affinity_scope: ContextVar[Optional[str]] = ContextVar(
"hermes_affinity_scope", default=None
)
def set_affinity_scope(scope: Optional[str]):
"""Publish the declared routing/affinity scope for this turn.
Returns the ContextVar token; pair with :func:`reset_affinity_scope`.
"""
return _affinity_scope.set(scope or None)
def reset_affinity_scope(token) -> None:
"""Restore the previous affinity scope (pair with ``set_affinity_scope``)."""
try:
_affinity_scope.reset(token)
except Exception:
_affinity_scope.set(None)
def get_affinity_scope() -> Optional[str]:
"""Return the declared routing/affinity scope, or ``None`` when unset."""
return _affinity_scope.get()
def set_conversation_context(conversation_id: Optional[str]):
"""Publish the active conversation id for ambient Portal tagging.
+48
View File
@@ -269,6 +269,49 @@ def _enable_happy_eyeballs(transport) -> None:
pool._network_backend = _HappyEyeballsSyncBackend()
def enable_happy_eyeballs_on_client(client) -> None:
"""Install the sync racing backend on every direct transport of a client.
Covers a ready-built ``httpx.Client`` (its default transport plus any
mounts), for callers that construct clients inline instead of going
through :func:`build_keepalive_http_client` — e.g. the Codex OAuth token
refresh / device-login / usage-probe clients in ``hermes_cli.auth``.
Proxy-backed transports (``httpcore.HTTPProxy`` / SOCKS pools) are left
untouched: with a proxy in play the TCP connect goes to the proxy host,
which is out of scope for the direct-transport racing added in #94388.
Async clients are also left untouched — httpcore's async backend already
performs RFC 8305 racing natively via
``anyio.connect_tcp(happy_eyeballs_delay=0.25)``.
Best-effort and hasattr-guarded like ``_enable_happy_eyeballs``; on an
incompatible httpx/httpcore this silently keeps the default backend.
"""
try:
import httpcore
proxy_pool_types = tuple(
t
for t in (
getattr(httpcore, "HTTPProxy", None),
getattr(httpcore, "SOCKSProxy", None),
)
if t is not None
)
except Exception:
return
transports = [getattr(client, "_transport", None)]
transports.extend((getattr(client, "_mounts", None) or {}).values())
for transport in transports:
pool = getattr(transport, "_pool", None)
if pool is None or not hasattr(pool, "_network_backend"):
continue
if proxy_pool_types and isinstance(pool, proxy_pool_types):
continue
pool._network_backend = _HappyEyeballsSyncBackend()
def _load_openai_cls() -> type:
"""Import and cache ``openai.OpenAI``."""
global _OPENAI_CLS_CACHE
@@ -421,6 +464,10 @@ def build_keepalive_http_client(
if proxy is None:
http_transport = transport_cls(verify=verify)
https_transport = transport_cls(verify=verify)
# Async transports need no explicit racing: httpcore's anyio
# backend already implements RFC 8305 natively
# (``anyio.connect_tcp(happy_eyeballs_delay=0.25)``), covered by
# tests/agent/test_codex_happy_eyeballs.py.
if not async_mode and _uses_codex_cloud_transport(base_url):
_enable_happy_eyeballs(http_transport)
_enable_happy_eyeballs(https_transport)
@@ -459,4 +506,5 @@ __all__ = [
"_get_proxy_from_env",
"_get_proxy_for_base_url",
"build_keepalive_http_client",
"enable_happy_eyeballs_on_client",
]
+2 -1
View File
@@ -1778,7 +1778,8 @@ def build_skills_system_prompt(
External skill directories (``skills.external_dirs`` in config.yaml) are
scanned alongside the local ``~/.hermes/skills/`` directory. External dirs
are read-only — they appear in the index but new skills are always created
in the local dir. Local skills take precedence when names collide.
in the local dir (or ``skills.create_dir`` when configured). Local skills
take precedence when names collide.
``compact_categories`` (e.g. from the coding posture — see
agent/coding_context.py) demotes whole categories to a names-only line in
+198 -4
View File
@@ -28,6 +28,36 @@ intentionally different — do not "deduplicate" them.
timestamp is stripped later by ``_cache_scope_from_session_id`` exactly as
before.
A host that mints one physical ``session_id`` per RESPONSE (Hermes Studio's
group chat, and ``POST /v1/responses`` with client-managed history, which
mints ``str(uuid4())`` per request) re-keys every conversation-affinity hint
Hermes sends — ``prompt_cache_key`` on both OpenAI-wire transports, plus the
OpenRouter/Nous sticky ``session_id`` and xAI's ``x-grok-conv-id`` through
``portal_tags`` (issue #96811). Those rows carry no lineage, so the walk
above correctly returns the physical id and the scope moves every reply.
Hermes must not infer the logical conversation from the id's SYNTAX (that
rule collides independent client-supplied ids and merges Studio members
truncated past its 96-character boundary — the #79017 failure class). The
host has to declare it, and one carrier already means exactly that:
``gateway_session_key`` — the "stable per-chat key" (``agent:main:telegram:
dm:123``) built by ``gateway.session.build_session_key`` from the
``X-Hermes-Session-Key`` header, which branching deliberately does NOT key
off. ``declared_conversation_scope()`` consumes it, and it wins over the
lineage walk because it is stable across rotation AND across per-response
ids. Two boundaries it must not cross:
- explicit fork children (``/branch``, delegate subagents, tool children)
share their parent's chat key but are separate conversations — the row's
fork markers keep them on their own scope (#79161);
- background-review forks run on a clone of the live runtime, so they are
excluded by ``_persist_disabled`` for the same reason.
The declared key is hashed into ``gwk_<sha256[:24]>`` before it becomes a
scope: unlike a session id it embeds platform/chat/user identifiers, and
this value leaves the process verbatim as OpenRouter's sticky ``session_id``
and xAI's ``x-grok-conv-id``.
The resolution is memoized per (agent, session_id): the lineage walk runs
once per transcript segment — NOT per API call — and re-runs only when
rotation actually changes ``agent.session_id`` (per the no-DB-on-the-hot-path
@@ -35,12 +65,15 @@ constraint recorded on #79017). Default installs compact in place and never
rotate, so they hit the memo forever and behave byte-identically to before.
"""
import hashlib
import logging
from typing import Any, Optional
logger = logging.getLogger(__name__)
_MEMO_ATTR = "_prompt_cache_scope_memo"
# Namespace for a scope resolved from a host-declared conversation key.
_DECLARED_SCOPE_PREFIX = "gwk_"
def _lineage_root(session_id: str, session_db: Any) -> Optional[str]:
@@ -64,12 +97,159 @@ def _lineage_root(session_id: str, session_db: Any) -> Optional[str]:
return None
def _agent_source(
agent: Any, session_id: str, session_db: Any, row_source: Optional[str] = None
) -> str:
"""The ``sessions.source`` this agent's conversation is recorded under.
Read from the agent's own row when it exists, because that is the value
the peer queries below match on.
``row_source`` is that value when the caller already has it — the single
identity read in :func:`declared_conversation_scope` — where ``""`` means
"the row was read and carries no source". ``None`` means "not read yet"
and keeps the original lookup, which is the path a ``SessionDB`` without
:meth:`~hermes_state.SessionDB.declared_scope_identity` still takes.
Before the row lands — this module resolves the first scope ahead of
``_ensure_db_session`` — it uses the SAME resolver persistence will use,
``run_agent._session_source_for_agent``, not ``agent.platform``. The two
diverge whenever ``HERMES_SESSION_SOURCE`` overrides the platform, and the
divergence is not a cosmetic one: the declared scope is non-``None``
immediately, so ``resolve_prompt_cache_scope`` memoizes it for this session
id and never re-resolves once the authoritative row appears. Both sides of
a ``/new`` would then read the platform domain, miss the boundary recorded
under the override, and hash the same scope.
"""
if row_source is None and session_id and session_db is not None:
try:
row = session_db.get_session(session_id)
except Exception:
logger.debug("declared-scope source lookup failed", exc_info=True)
row = None
row_source = str(row.get("source") or "").strip() if row else ""
if row_source:
return row_source
platform = getattr(agent, "platform", None)
try:
# Imported lazily: run_agent imports this module, and this is the
# single owner of the source a session row is created with.
from run_agent import _session_source_for_agent
source = str(_session_source_for_agent(platform) or "").strip()
if source:
return source
except Exception:
logger.debug("declared-scope source authority unavailable", exc_info=True)
return str(platform or "").strip()
def _conversation_generation(session_key: str, source: str, session_db: Any) -> str:
"""Return the durable generation for *session_key*'s current conversation.
The declared key names a chat and deliberately survives `/new` and policy
resets. Hashing it alone would therefore reuse one affinity scope across
distinct conversations, violating the #79017/#86733 contract: warm across
compression, cold across a conversation boundary.
``SessionDB.latest_conversation_boundary`` reads the monotonic
``conversation_generations`` counter for ``(source, session_key)``. The
counter advances in the same transaction that records an
``_RESET_END_REASONS`` boundary. It is independent of prunable session rows
and wall-clock time, so deletion, bulk pruning, and clock rollback cannot
reissue an old generation. Compression continues the current conversation
and does not advance it.
This lookup runs on the memoized resolution path, not once per API call.
Return ``""`` when the key has never reset or the DB exposes no generation.
"""
reader = getattr(session_db, "latest_conversation_boundary", None)
if not callable(reader):
return ""
generation = reader(session_key, source)
if generation is None:
return ""
return str(int(generation))
def declared_conversation_scope(agent: Any) -> Optional[str]:
"""Return the host-declared logical conversation scope, or None.
Resolved from ``agent._gateway_session_key`` (the ``X-Hermes-Session-Key``
/``build_session_key`` per-chat key) qualified by the conversation
generation currently live on it (:func:`_conversation_generation`), hashed
together into ``gwk_<sha256[:24]>`` so no platform/chat/user identifier
reaches a provider on the wire and the value stays inside every caller's
length/charset budget.
The key alone would outlive the conversation — it survives ``/new`` and the
idle/daily policy resets by design — so the generation is what makes this
carrier legal: stable across a host's per-response physical ids, and cold
on every conversation replacement.
None — meaning "fall back to the physical-id scope" — when no key was
declared, when this agent is a background-review fork (``_persist_disabled``:
it clones the live runtime, including the key), when the session row is an
explicit fork child (``/branch``, delegate, tool), and on any DB error
during either lookup.
"""
key = str(getattr(agent, "_gateway_session_key", "") or "").strip()
if not key:
return None
if getattr(agent, "_persist_disabled", False):
return None
sid = str(getattr(agent, "session_id", None) or "")
db = getattr(agent, "_session_db", None)
generation = ""
row_source: Optional[str] = None
if sid and db is not None:
try:
# One read for both halves of the row's identity: the fork verdict
# and the source the peer queries match on live on the same
# ``sessions`` row, and asking for them separately read it twice
# per resolution (@teknium1 on #98811). A SessionDB without the
# combined view keeps the original call, so nothing that predates
# it — including the doubles that certify the fail-closed contract
# below — changes behaviour.
identity = getattr(db, "declared_scope_identity", None)
if callable(identity):
is_fork, row_source = identity(sid)
else:
is_fork = db.is_explicit_fork_child(sid)
if is_fork:
return None
except Exception:
# Degrade to the physical-id scope rather than risk merging a
# fork onto its parent's key on a transient DB failure. The
# source read is inside this same guard for the same reason: it
# was always the second half of a read that had already failed
# closed here.
logger.debug("declared-scope fork check failed", exc_info=True)
return None
source = _agent_source(agent, sid, db, row_source)
if db is not None:
try:
generation = _conversation_generation(key, source, db)
except Exception:
# Same fail-closed rule as the fork check: an unqualified key
# spans /new, so degrade to the physical-id scope instead.
logger.debug("declared-scope generation read failed", exc_info=True)
return None
# The carrier is the SAME identity tuple the peer queries use: two hosts
# may legally declare the same key string under different sources, and the
# scope leaves this process as a routing key, so it must not collapse them.
carrier = f"{source}|{key}|{generation}"
digest = hashlib.sha256(carrier.encode("utf-8", errors="replace")).hexdigest()[:24]
return f"{_DECLARED_SCOPE_PREFIX}{digest}"
def resolve_prompt_cache_scope(agent: Any) -> str:
"""Resolve the rotation-stable cache-scope id for *agent*'s conversation.
Returns the compression-lineage ROOT of ``agent.session_id`` (the
physical id itself when the session has no compression ancestry, no DB
is attached, or the walk fails). The result is memoized on the agent
Returns the host-declared conversation scope when one applies
(``declared_conversation_scope``), else the compression-lineage ROOT of
``agent.session_id`` (the physical id itself when the session has no
compression ancestry, no DB is attached, or the walk fails). The result is memoized on the agent
keyed by the current session id, so the DB walk happens once per
transcript segment rather than once per API call.
"""
@@ -84,7 +264,12 @@ def resolve_prompt_cache_scope(agent: Any) -> str:
memo = getattr(agent, _MEMO_ATTR, None)
if isinstance(memo, tuple) and len(memo) == 2 and memo[0] == key:
return memo[1]
root = _lineage_root(sid, db) if db is not None else None
# A declared conversation key outranks the lineage walk: it is stable
# across compression rotation AND across a host's per-response ids, which
# the walk cannot see (#96811).
root = declared_conversation_scope(agent) or (
_lineage_root(sid, db) if db is not None else None
)
scope = root or sid
# Memoize on a successful walk, or when there is no DB to consult at all,
# or when the agent will never persist a row (background-review forks set
@@ -108,6 +293,15 @@ def resolve_prompt_cache_scope(agent: Any) -> str:
return scope
def declared_conversation_scope_safe(agent: Any) -> Optional[str]:
"""Never-raising variant of :func:`declared_conversation_scope`."""
try:
return declared_conversation_scope(agent)
except Exception:
logger.debug("declared conversation scope resolution failed", exc_info=True)
return None
def resolve_prompt_cache_scope_safe(agent: Any) -> Optional[str]:
"""Never-raising variant of :func:`resolve_prompt_cache_scope`.
+64 -1
View File
@@ -158,8 +158,12 @@ _ENV_ASSIGN_RE = re.compile(
# Lowercase env names: only underscore-boundary forms (``openai_key=…``,
# ``FAL_KEY=…``, ``db_pw=…``) — NOT bare ``password=``/``token=``/``secret=``,
# which appear in prose, URLs, and form bodies (issue #77484).
# Anchor each attempt to the start of an identifier run. Without the
# lookbehind, ``re.sub`` retries the greedy ``[a-z0-9_]+`` prefix at every byte
# of a long non-matching opaque payload, making strict compaction redaction
# quadratic while holding the GIL (#99255).
_ENV_ASSIGN_LOWER_RE = re.compile(
rf"([a-z0-9_]+(?:_|^)(?:key|pass|pw|token|secret|password|passwd|credential|auth)(?=[^a-z0-9_]|$))\s*=\s*(['\"]?)(\S+)\2",
rf"(?<![a-z0-9_])([a-z0-9_]+(?:_|^)(?:key|pass|pw|token|secret|password|passwd|credential|auth)(?=[^a-z0-9_]|$))\s*=\s*(['\"]?)(\S+)\2",
re.IGNORECASE,
)
@@ -200,7 +204,14 @@ _ENV_LOOKUP_VALUE_RE = re.compile(
# ``(?:[A-Za-z0-9_\-]+\.)+`` (exponential backtracking on long dotted runs).
# The ``*`` runs bordering {_SECRET_CFG_NAMES} must stay backtrackable
# (secret words are matchable by the class, e.g. ``app.api.key=…``).
# The lookbehind anchors each attempt to the start of a key run: without it,
# ``re.sub`` retries the backtrackable ``*`` prefix at every byte of a long
# non-matching dotted run, making the sub quadratic whenever the text contains
# a secret keyword anywhere (the ``_CFG_SECRET_WORD_RE`` pre-gate only skips
# secret-free text). Match set is unchanged — any match starting mid-run
# implies a leftmost match starting at the run start (#99255).
_CFG_DOTTED_RE = re.compile(
rf"(?<![A-Za-z0-9_.\-])"
rf"([A-Za-z0-9_\-]++\.[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*+"
rf"|[A-Za-z0-9_.\-]*{_SECRET_CFG_NAMES}[A-Za-z0-9_.\-]*\.[A-Za-z0-9_.\-]++)"
rf"={_CFG_VALUE}",
@@ -259,6 +270,16 @@ _KEY_KEYWORD_RE = re.compile(
re.IGNORECASE,
)
# Key names that are credential-specific even when their values are short or
# human-readable. Bare ``token`` / ``key`` are intentionally absent: those
# words also describe model limits, tensor names, cache keys, and other public
# technical values. Their assignments are gated on value shape below.
_STRONG_KEY_KEYWORD_RE = re.compile(
r"(?:api|auth|access|refresh|session|id|bearer)[ _.\\-]?(?:key|token)"
r"|key[ _.\\-]?material|secret|passwd|password|pass|pw|credential|auth|bearer",
re.IGNORECASE,
)
def _is_word_start(s: str, i: int) -> bool:
"""True if position ``i`` in ``s`` begins a word (not mid-word)."""
@@ -315,6 +336,42 @@ def _key_has_secret_keyword(key: str) -> bool:
return True
return False
def _key_has_strong_secret_keyword(key: str) -> bool:
"""Return whether ``key`` names an unambiguously credential-bearing field."""
for match in _STRONG_KEY_KEYWORD_RE.finditer(key):
if _is_word_start(key, match.start()) and _is_word_end(key, match.end()):
return True
return False
def _looks_like_opaque_credential(value: str) -> bool:
"""Return whether an ambiguous token/key value has credential-like shape.
Known vendor prefixes and JWTs have dedicated redactors. This catches the
remaining opaque family without treating short technical scalars such as
``CPU``, ``local``, or training captions as secrets merely because their
key contains ``token`` or ``key``.
"""
if value == "***" or value.startswith("«redacted:"):
return True
if len(value) >= 16 and re.fullmatch(r"[A-Fa-f0-9]+", value):
return True
if len(value) >= 20 and re.fullmatch(r"[A-Za-z0-9_./+=-]+", value):
return True
if len(value) < 12:
return False
classes = sum(
bool(re.search(pattern, value))
for pattern in (r"[a-z]", r"[A-Z]", r"[0-9]")
)
return classes >= 2
def _assignment_value_requires_redaction(key: str, value: str) -> bool:
"""Apply value-aware gating to key-name-only assignment matches."""
return _key_has_strong_secret_keyword(key) or _looks_like_opaque_credential(value)
# JSON field patterns: "apiKey": "value", "token": "value", etc.
_JSON_KEY_NAMES = r"(?:api_?[Kk]ey|token|secret|password|access_token|refresh_token|auth_token|bearer|secret_value|raw_secret|secret_input|key_material)"
_JSON_FIELD_RE = re.compile(
@@ -859,6 +916,8 @@ def redact_sensitive_text(
# embedded matching inside the helper.
if not _key_has_secret_keyword(name):
return m.group(0)
if not _assignment_value_requires_redaction(name, value):
return m.group(0)
return f"{name}={quote}{_mask_token(value)}{quote}"
text = _ENV_ASSIGN_RE.sub(_redact_env, text)
# Lowercase env names (``openai_key=…``). Skip URLs — the query
@@ -894,6 +953,8 @@ def redact_sensitive_text(
# not a leaked secret value.
if _ENV_LOOKUP_VALUE_RE.match(value):
return m.group(0)
if not _assignment_value_requires_redaction(key, value):
return m.group(0)
return f'{key}: "{_mask_token(value)}"'
text = _JSON_FIELD_RE.sub(_redact_json, text)
@@ -913,6 +974,8 @@ def redact_sensitive_text(
# document text, not credentials (nearai/ironclaw#6129).
if not _key_has_secret_keyword(key):
return m.group(0)
if not _assignment_value_requires_redaction(key, value):
return m.group(0)
return f"{key}{sep}{_mask_token(value)}"
text = _YAML_ASSIGN_RE.sub(_redact_yaml, text)
+291
View File
@@ -0,0 +1,291 @@
"""Idle deferral for background reviews on the managed local runtime.
The post-turn review fork replays the whole conversation on the review
runtime. On a cloud provider that costs seconds and runs concurrently
with whatever the user does next. When the review runtime IS the managed
llama-server, the same fork monopolizes the GPU the user's next prompt
needs, for minutes — and the next live turn cancels it, so an active
session tends to pay the decode cost AND lose the learning.
This module keeps the decision to learn exactly where it was (turn end,
nudge intervals, full-strength model, full transcript) and moves only
the execution moment: reviews bound for the managed local endpoint are
queued and dispatched when the machine is quiet. Everything else runs
immediately, as before.
Policy (auxiliary.background_review.defer):
auto (default) — defer exactly when the resolved review runtime
targets the managed local server.
never — old behavior everywhere.
Explicit /refine (focus set) never defers: an explicit ask runs now,
matching its bypass of the enabled gate.
Queue semantics:
- One slot per session, newest snapshot wins. A review replays the whole
conversation, so a newer snapshot strictly supersedes an older one —
coalescing is deduplication, not loss.
- Preempted (cancelled-by-live-turn) reviews are requeued by the spawn
wrapper observing the run token's cancel flag, not killed-and-forgotten.
- Aged-out events (defer_max_age_s, default 30 min) dispatch regardless
of idleness — deferral may delay learning, never lose it.
- In-memory, best-effort: dropped on process exit, the same durability
contract the immediate daemon-thread fork always had.
Idle truth comes from the supervisor's /slots (machine-level: it sees
every client of the managed server, including other Hermes profiles) and
must hold for a settle window so a review is not launched into the gap
between two quick prompts. Local in-process turn liveness is tracked via
note_turn_started/note_turn_finished from run_conversation.
"""
from __future__ import annotations
import json
import logging
import threading
import time
import urllib.request
from typing import Any, Callable, Dict, List, Optional
logger = logging.getLogger(__name__)
# Sustained-quiet window before dispatch. Long enough that "typed two
# prompts back to back" does not look idle; short enough that walking
# away for coffee runs the queue.
_IDLE_SETTLE_S = 15.0
# Poll cadence while the queue is non-empty. The thread parks when empty.
_POLL_INTERVAL_S = 5.0
# Age at which a queued review dispatches regardless of idleness.
_MAX_AGE_DEFAULT_S = 30.0 * 60.0
def defer_mode(task_cfg: Optional[Dict[str, Any]]) -> str:
"""'auto' (default) or 'never' from auxiliary.background_review.defer."""
raw = str((task_cfg or {}).get("defer", "auto")).strip().lower()
return raw if raw in ("auto", "never") else "auto"
def defer_max_age_s(task_cfg: Optional[Dict[str, Any]]) -> float:
raw = (task_cfg or {}).get("defer_max_age_s", _MAX_AGE_DEFAULT_S)
try:
value = float(raw)
except (TypeError, ValueError):
return _MAX_AGE_DEFAULT_S
return value if value > 0 else _MAX_AGE_DEFAULT_S
def review_targets_managed_local(agent: Any,
task_cfg: Optional[Dict[str, Any]]) -> bool:
"""Would this review fork decode on the llama-server WE manage?
Resolves the review runtime the same way the fork itself will and
exact-matches its netloc against the supervisor state file — the
matcher that cannot false-positive on external local servers. Any
failure reads False: immediate spawn is always the safe default.
Order matters: the netloc probe (one TTL-cached state-file read)
runs FIRST, so machines with no managed server — every cloud-only
install — return False without resolving the review runtime at all.
This wrapper runs on the turn's tail; runtime resolution belongs on
that path only when a managed server actually exists.
"""
try:
from agent.auxiliary_client import (
_is_managed_local_endpoint,
_managed_local_netloc,
)
if not _managed_local_netloc():
return False
from agent.background_review import _resolve_review_runtime
runtime = _resolve_review_runtime(agent, task_cfg)
return _is_managed_local_endpoint(runtime.get("base_url"))
except Exception: # noqa: BLE001
return False
class _PendingReview:
__slots__ = ("agent", "kwargs", "enqueued_at", "session_key")
def __init__(self, agent: Any, session_key: str, kwargs: Dict[str, Any]):
self.agent = agent
self.session_key = session_key
self.kwargs = kwargs
self.enqueued_at = time.monotonic()
class ReviewIdleQueue:
"""Session-coalescing queue + idle-gated dispatcher thread."""
def __init__(self) -> None:
self._lock = threading.Lock()
self._pending: Dict[str, _PendingReview] = {}
self._wake = threading.Event()
self._thread: Optional[threading.Thread] = None
self._live_turns = 0
self._quiet_since: Optional[float] = None
# Test seams — replaced by unit tests, never in production.
self._now: Callable[[], float] = time.monotonic
self._server_idle: Callable[[], bool] = _managed_server_idle
# ── turn liveness (this process) ────────────────────────────
def note_turn_started(self) -> None:
with self._lock:
self._live_turns += 1
self._quiet_since = None
def note_turn_finished(self) -> None:
with self._lock:
self._live_turns = max(0, self._live_turns - 1)
if self._live_turns == 0:
self._quiet_since = self._now()
self._wake.set()
# ── queue ────────────────────────────────────────────────────
def enqueue(self, agent: Any, session_key: str,
kwargs: Dict[str, Any]) -> None:
"""Add (or replace — newest snapshot wins) a session's pending review."""
with self._lock:
existing = self._pending.get(session_key)
item = _PendingReview(agent, session_key, kwargs)
# Stamp through the queue's clock (test seam); keep the ORIGINAL
# enqueue time on coalesce so a busy session cannot push its
# review's age-out forever.
item.enqueued_at = (existing.enqueued_at if existing is not None
else self._now())
self._pending[session_key] = item
self._ensure_thread()
self._wake.set()
logger.info("Background review deferred (session=%s, queued=%d)",
session_key[-12:], len(self._pending))
def pending_count(self) -> int:
with self._lock:
return len(self._pending)
# ── dispatcher ───────────────────────────────────────────────
def _ensure_thread(self) -> None:
with self._lock:
if self._thread is None or not self._thread.is_alive():
self._thread = threading.Thread(
target=self._run, daemon=True, name="bg-review-idle-queue")
self._thread.start()
def _quiet_for(self) -> float:
"""Seconds this process has been turn-free (0 while a turn runs)."""
with self._lock:
if self._live_turns > 0 or self._quiet_since is None:
return 0.0
return self._now() - self._quiet_since
def _pop_dispatchable(self) -> Optional[_PendingReview]:
"""Oldest aged-out item, else any item once quiet+idle hold."""
with self._lock:
if not self._pending:
return None
items = sorted(self._pending.values(),
key=lambda p: p.enqueued_at)
aged = [p for p in items
if self._now() - p.enqueued_at
>= defer_max_age_s(p.kwargs.get("task_cfg"))]
candidate = aged[0] if aged else None
if candidate is None:
if self._quiet_for() < _IDLE_SETTLE_S:
return None
if not self._server_idle():
return None
with self._lock:
if not self._pending:
return None
candidate = min(self._pending.values(),
key=lambda p: p.enqueued_at)
with self._lock:
return self._pending.pop(candidate.session_key, None)
def _run(self) -> None:
while True:
self._wake.wait()
with self._lock:
if not self._pending:
self._wake.clear()
continue
item = None
try:
item = self._pop_dispatchable()
if item is not None:
if not self._still_enabled(item):
logger.info(
"Deferred background review dropped: reviews "
"were disabled while it was queued (session=%s)",
item.session_key[-12:])
continue
logger.info(
"Dispatching deferred background review "
"(session=%s, waited=%.0fs, queued=%d)",
item.session_key[-12:],
self._now() - item.enqueued_at,
self.pending_count())
item.agent._spawn_background_review_now(**item.kwargs)
except Exception: # noqa: BLE001 — dispatcher must survive anything
logger.warning("Deferred review dispatch failed",
exc_info=True)
if item is None:
time.sleep(_POLL_INTERVAL_S)
@staticmethod
def _still_enabled(item: _PendingReview) -> bool:
"""Re-check the enabled gate at DISPATCH time.
The entry wrapper gates at enqueue time, but minutes may pass in
the queue — a user who sets background_review.enabled: false while
a review waits means it, and the dispatch must not resurrect it.
Fail-open like the gate itself (a broken config never silently
disables reviews)."""
try:
from agent.background_review import load_background_review_settings
enabled, _ = load_background_review_settings()
return enabled
except Exception: # noqa: BLE001
return True
def _managed_server_idle() -> bool:
"""Machine-level idle: no processing slot on any loaded model of the
managed router. Unreachable/no state file reads idle (nothing to
contend with). One /models + one /slots call per loaded model."""
try:
from hermes_cli.local_runtime.supervisor import state_path
state = json.loads(state_path().read_text(encoding="utf-8"))
base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
key = str(state.get("api_key", ""))
if not base:
return True
headers = {"Authorization": f"Bearer {key}"}
req = urllib.request.Request(f"{base}/models", headers=headers)
with urllib.request.urlopen(req, timeout=3) as r:
models = json.loads(r.read())
loaded = [m["id"] for m in models.get("data", [])
if (m.get("status") or {}).get("value") in ("loaded", "ready")]
from urllib.parse import quote
for mid in loaded:
req = urllib.request.Request(f"{base}/slots?model={quote(mid)}",
headers=headers)
with urllib.request.urlopen(req, timeout=3) as r:
slots = json.loads(r.read())
if any(s.get("is_processing") for s in slots
if isinstance(s, dict)):
return False
return True
except Exception: # noqa: BLE001
return True
# Module singleton — one queue per process, like the load-progress watcher.
QUEUE = ReviewIdleQueue()
+73 -4
View File
@@ -613,11 +613,76 @@ def get_external_skills_dirs() -> List[Path]:
return result
def get_skill_create_dir() -> Optional[Path]:
"""Return the configured ``skills.create_dir``, or ``None`` when unset.
When set, agent-created skills (``skill_manage`` action=create) land in
this directory instead of the profile-local ``~/.hermes/skills/``, and
every user-facing instruction string that names the creation path renders
this directory instead of the default.
The entry is expanded (``~`` and ``${VAR}``); relative paths resolve
against HERMES_HOME. A value that resolves to the local skills dir is
treated as unset (that is already the default behaviour). The directory
does NOT need to exist yet — skill creation mkdirs it on first write.
"""
parsed = _load_raw_config()
if not parsed:
return None
skills_cfg = parsed.get("skills")
if not isinstance(skills_cfg, dict):
return None
raw = skills_cfg.get("create_dir")
if not raw or not isinstance(raw, (str, os.PathLike)):
return None
entry = str(raw).strip()
if not entry:
return None
from hermes_constants import get_hermes_home
expanded = os.path.expanduser(os.path.expandvars(entry))
p = Path(expanded)
if not p.is_absolute():
p = get_hermes_home() / p
try:
resolved = p.resolve()
except OSError:
resolved = p
try:
if resolved == get_skills_dir().resolve():
return None
except OSError:
pass
return resolved
def display_skill_create_dir() -> str:
"""User-facing display string for where new skills are created.
Renders the configured ``skills.create_dir`` (with ``~/`` shorthand when
under the user's home) or the default ``<home>/skills/`` path. Used by
instruction text (tool schema descriptions, prompts, docs strings) so a
configured creation dir changes every instruction that names the path.
"""
from hermes_constants import display_hermes_home
create_dir = get_skill_create_dir()
if create_dir is None:
return f"{display_hermes_home()}/skills/"
try:
return "~/" + create_dir.relative_to(Path.home()).as_posix() + "/"
except ValueError:
return create_dir.as_posix() + "/"
def get_all_skills_dirs() -> List[Path]:
"""Return all skill directories: local ``~/.hermes/skills/`` first, then external.
The local dir is always first (and always included even if it doesn't exist
yet — callers handle that). External dirs follow in config order.
yet — callers handle that). When ``skills.create_dir`` is configured, it
follows immediately after the local dir (so agent-created skills are
discovered, trusted, and modifiable). External dirs follow in config order.
NOTE: trusted project-local dirs (``./.hermes/skills`` at the git root) are
NOT part of this list — they have *higher* precedence than the local dir,
@@ -626,7 +691,12 @@ def get_all_skills_dirs() -> List[Path]:
precedence-ordered list.
"""
dirs = [get_skills_dir()]
dirs.extend(get_external_skills_dirs())
create_dir = get_skill_create_dir()
if create_dir is not None and create_dir.is_dir():
dirs.append(create_dir)
for d in get_external_skills_dirs():
if d not in dirs:
dirs.append(d)
return dirs
@@ -810,8 +880,7 @@ def get_scan_ordered_skills_dirs() -> List[Path]:
priority over profile-local and external ones.
"""
dirs = list(get_project_skills_dirs())
dirs.append(get_skills_dir())
dirs.extend(get_external_skills_dirs())
dirs.extend(get_all_skills_dirs())
return dirs
+9 -1
View File
@@ -892,8 +892,16 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None)
# (developing Hermes). Every other surface (desktop chat panel,
# gateway daemons) self-spawns into the install tree, where the
# fallback would inject this repo's contributor AGENTS.md (#64590).
context_cwd = resolve_context_cwd()
if getattr(agent, "_context_cwd_is_launch_artifact", False):
# Desktop session creation pins the backend launch directory so
# tools have a deterministic cwd even when the user picked no
# workspace. Preserve that tool routing, but let context discovery
# see it as the fallback it really is. The install-tree guard can
# then reject Hermes's bundled contributor AGENTS.md (#97448).
context_cwd = None
context_files_prompt = _r.build_context_files_prompt(
cwd=resolve_context_cwd(), skip_soul=_soul_loaded,
cwd=context_cwd, skip_soul=_soul_loaded,
context_length=_ctx_len,
allow_install_tree_fallback=agent.platform in ("cli", "tui"),
home_override=_agent_home(agent))
+11 -1
View File
@@ -84,7 +84,11 @@ class AnthropicTransport(ProviderTransport):
to OpenAI finish_reason, and collects reasoning_details in provider_data.
"""
import json
from agent.anthropic_adapter import _to_plain_data, _sanitize_replay_block
from agent.anthropic_adapter import (
_OAUTH_TOOL_NAME_REVERSE_ALIASES,
_sanitize_replay_block,
_to_plain_data,
)
from agent.transports.types import ToolCall
strip_tool_prefix = kwargs.get("strip_tool_prefix", False)
@@ -151,6 +155,12 @@ class AnthropicTransport(ProviderTransport):
name = single
elif _tool_registry.get_entry(bare):
name = bare
elif bare in _OAUTH_TOOL_NAME_REVERSE_ALIASES:
# OAuth wire alias (e.g. chat_history_lookup ->
# session_search, #65365). Checked LAST so a real
# tool actually registered under the wire name
# still wins — same GH-25255 precedence.
name = _OAUTH_TOOL_NAME_REVERSE_ALIASES[bare]
tool_calls.append(
ToolCall(
id=block.id,
+7
View File
@@ -107,12 +107,19 @@ class BedrockTransport(ProviderTransport):
reasoning = getattr(msg, "reasoning", None) or getattr(msg, "reasoning_content", None)
provider_data = {}
if getattr(msg, "reasoning_details", None):
provider_data["reasoning_details"] = msg.reasoning_details
if getattr(msg, "bedrock_content_blocks", None):
provider_data["bedrock_content_blocks"] = msg.bedrock_content_blocks
return NormalizedResponse(
content=msg.content,
tool_calls=tool_calls,
finish_reason=finish_reason,
reasoning=reasoning,
usage=usage,
provider_data=provider_data or None,
)
def validate_response(self, response: Any) -> bool:
+88
View File
@@ -27,6 +27,61 @@ from agent.prompt_builder import DEVELOPER_ROLE_MODELS
from agent.transports.base import ProviderTransport
from agent.transports.types import NormalizedResponse, ToolCall, Usage
# xAI's chat-completions API reserves the function name ``tool_search`` for
# its own server-side tool and rejects any request declaring a client
# function with that name (HTTP 400 "The function name tool_search is
# reserved for the tool_search tool", #95003). The Tool Search bridge
# (tools/tool_search.py) assembles its client-side discovery tool under the
# same literal name for every provider, so Grok providers are unusable
# whenever the bridge is active. Mirror the web_search treatment in
# transports/codex.py (_rename_client_web_search_for_xai): alias the wire
# declaration and map the alias back in normalize_response. The alias value
# matches _CODEX_TOOL_SEARCH_ALIAS from the Codex-side fix for the same
# reserved-name class (#83122) so the two transports stay consistent.
_XAI_TOOL_SEARCH_ALIAS = "hermes_tool_search"
def _rename_tool_search_bridge_for_xai(
tools: list[dict[str, Any]],
) -> tuple[list[dict[str, Any]], dict[str, str]]:
"""Rename the client ``tool_search`` bridge declaration to a wire alias.
Only the wire name changes: descriptions, schemas, and the other two
bridge names (``tool_describe`` / ``tool_call`` — not reserved by xAI)
pass through untouched. Returns ``(rewritten_tools, alias_map)`` where
``alias_map`` maps each alias THIS request emits back to the original
name; the caller stashes it on the transport so ``normalize_response``
only reverses aliases that were actually sent. If a real tool already
occupies ``hermes_tool_search``, the bridge takes a ``_2``/``_3``
suffix instead of duplicating a wire name.
"""
rewritten: list[dict[str, Any]] = []
alias_map: dict[str, str] = {}
taken = {
(tool.get("function") or {}).get("name")
for tool in tools
if isinstance(tool, dict)
}
taken.discard(None)
for tool in tools:
if (
isinstance(tool, dict)
and (tool.get("function") or {}).get("name") == "tool_search"
):
alias = _XAI_TOOL_SEARCH_ALIAS
suffix = 2
while alias in taken:
alias = f"{_XAI_TOOL_SEARCH_ALIAS}_{suffix}"
suffix += 1
taken.add(alias)
alias_map[alias] = "tool_search"
aliased = dict(tool)
aliased["function"] = {**tool["function"], "name": alias}
rewritten.append(aliased)
else:
rewritten.append(tool)
return rewritten, alias_map
def _static_prompt_instructions(messages: list[dict[str, Any]]) -> str:
"""Return the stable system/developer prefix used for cache routing.
@@ -277,6 +332,13 @@ class ChatCompletionsTransport(ProviderTransport):
The default path for OpenAI-compatible providers.
"""
# Wire-alias provenance of the most recent request built for this
# transport: ``{alias_sent_on_wire: original_tool_name}``. ``None``
# means no request recorded provenance (normalize-only call sites) —
# fall back to the static alias constant. An empty dict means the last
# request emitted no aliases, so no reverse rewrite may run (#95003).
_last_wire_aliases: dict[str, str] | None = None
@property
def api_mode(self) -> str:
return "chat_completions"
@@ -316,6 +378,10 @@ class ChatCompletionsTransport(ProviderTransport):
gateways (e.g. opencode-go, codex.nekos.me) reject with
``Extra inputs are not permitted, field: 'messages[N]._empty_recovery_synthetic'``,
which then poisons every subsequent request in the session.
- Provider-specific ordered replay sidecars --
``anthropic_content_blocks`` and ``bedrock_content_blocks`` are
durable-history data for their native transports, not part of the
Chat Completions schema. They must not cross a provider boundary.
"""
strip_extra_content = not _model_consumes_thought_signature(
kwargs.get("model")
@@ -330,7 +396,10 @@ class ChatCompletionsTransport(ProviderTransport):
or "tool_name" in msg
or "effect_disposition" in msg
or "timestamp" in msg # #47868 — strict providers reject this
or "platform_message_id" in msg # gateway dedup id (persistence-only)
or "api_content" in msg # persist-what-you-send sidecar
or "anthropic_content_blocks" in msg
or "bedrock_content_blocks" in msg
):
needs_sanitize = True
break
@@ -401,7 +470,10 @@ class ChatCompletionsTransport(ProviderTransport):
or "tool_name" in msg
or "effect_disposition" in msg
or "timestamp" in msg # #47868 — leak into strict providers
or "platform_message_id" in msg # gateway dedup id (persistence-only)
or "api_content" in msg # persist-what-you-send sidecar
or "anthropic_content_blocks" in msg
or "bedrock_content_blocks" in msg
):
out_msg = mutable_msg()
out_msg.pop("codex_reasoning_items", None)
@@ -409,7 +481,10 @@ class ChatCompletionsTransport(ProviderTransport):
out_msg.pop("tool_name", None)
out_msg.pop("effect_disposition", None)
out_msg.pop("timestamp", None) # #47868 — leak into strict providers
out_msg.pop("platform_message_id", None) # gateway dedup id
out_msg.pop("api_content", None) # persist-what-you-send sidecar
out_msg.pop("anthropic_content_blocks", None)
out_msg.pop("bedrock_content_blocks", None)
# Drop all Hermes-internal scaffolding markers (``_``-prefixed).
@@ -925,6 +1000,19 @@ class ChatCompletionsTransport(ProviderTransport):
# preserve an explicit blank name for Hermes's recovery path.
if tc_function is None or function_name is None:
continue
# Map THIS request's wire aliases back before dispatch.
# Request-local provenance: when the paired request recorded
# its alias map, only those aliases are reversed — a real
# user/plugin/MCP tool that happens to be named
# ``hermes_tool_search`` dispatches as itself when no alias
# was emitted. The static-constant fallback covers
# normalize-only call sites with no recorded request.
_alias_map = self._last_wire_aliases
if _alias_map is None:
if function_name == _XAI_TOOL_SEARCH_ALIAS:
function_name = "tool_search"
elif function_name in _alias_map:
function_name = _alias_map[function_name]
function_arguments = getattr(tc_function, "arguments", None)
# Preserve provider-specific extras on the tool call.
# Gemini 3 thinking models attach extra_content with
+108 -16
View File
@@ -9,7 +9,7 @@ import hashlib
import json
import logging
import re
from typing import Any, Dict, List, Optional
from typing import Any, Dict, List, Optional, Tuple
logger = logging.getLogger(__name__)
@@ -70,10 +70,35 @@ _XAI_CLIENT_WEB_SEARCH_ALIAS = "hermes_web_search"
# rename on the wire (hermes_<name>), map back in normalize_response so
# Hermes dispatch is unaffected.
_OPENCODE_RESERVED_TOOL_NAMES = ("web_search", "search_files")
# xAI reserves ``tool_search`` server-side for Grok's own native Tool Search
# and rejects *any* client function declared with that name:
# HTTP 400 {"code":"invalid-argument","error":"The function name
# tool_search is reserved for the tool_search tool"}
# Hermes's progressive-disclosure bridge registers exactly that literal
# (``tools.tool_search.TOOL_SEARCH_NAME``), and assembly is not provider
# gated, so with ``tools.tool_search.enabled: auto`` a grok turn dies the
# moment the catalog crosses the threshold — mid-session, and only for
# sessions large enough to activate the bridge. Same treatment as the two
# collisions above. ``tool_describe`` / ``tool_call`` are not reserved.
# Refs #95003.
_XAI_RESERVED_TOOL_NAMES = ("tool_search",)
_RESERVED_TOOL_ALIAS_PREFIX = "hermes_"
_RESERVED_ALIAS_TO_NAME = {
f"{_RESERVED_TOOL_ALIAS_PREFIX}{name}": name
for name in _OPENCODE_RESERVED_TOOL_NAMES
for name in (*_OPENCODE_RESERVED_TOOL_NAMES, *_XAI_RESERVED_TOOL_NAMES)
}
# Legacy reverse map used ONLY when normalize_response runs on a transport
# instance that never built a request (normalize-only call sites / tests).
# Production requests carry request-local provenance instead — see
# ``_last_wire_aliases`` — so a real user/plugin/MCP tool that happens to be
# named ``hermes_tool_search`` is never silently rewritten to ``tool_search``
# unless THIS request actually emitted that alias (#95003 review contract).
_LEGACY_ALIAS_FALLBACK = {
**_RESERVED_ALIAS_TO_NAME,
"hermes_web_search": "web_search",
}
@@ -100,17 +125,44 @@ def _is_opencode_responses_backend(params: Dict[str, Any]) -> bool:
return False
def _rename_reserved_tools_for_opencode(response_tools: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
"""Alias OpenCode-reserved client function names on the wire."""
def _alias_reserved_tools(
response_tools: List[Dict[str, Any]],
reserved_names: Tuple[str, ...],
) -> Tuple[List[Dict[str, Any]], Dict[str, str]]:
"""Alias provider-reserved client function names on the wire.
Single owner for every reserved-name collision on this transport.
Returns ``(rewritten_tools, alias_map)`` where ``alias_map`` maps each
wire alias emitted by THIS request back to the original tool name.
The caller stashes the map for ``normalize_response`` so the reverse
rewrite only ever applies to aliases this request actually sent —
a legitimate user/plugin/MCP tool already named ``hermes_<x>`` is
neither shadowed (the alias picks a ``_2``/``_3`` suffix instead of
duplicating a wire name) nor mis-dispatched on the response path.
"""
rewritten: List[Dict[str, Any]] = []
alias_map: Dict[str, str] = {}
taken = {
tool.get("name")
for tool in response_tools
if isinstance(tool, dict) and tool.get("name")
}
for tool in response_tools:
if isinstance(tool, dict) and tool.get("name") in _OPENCODE_RESERVED_TOOL_NAMES:
if isinstance(tool, dict) and tool.get("name") in reserved_names:
base = f"{_RESERVED_TOOL_ALIAS_PREFIX}{tool['name']}"
alias = base
suffix = 2
while alias in taken:
alias = f"{base}_{suffix}"
suffix += 1
taken.add(alias)
alias_map[alias] = tool["name"]
aliased = dict(tool)
aliased["name"] = f"{_RESERVED_TOOL_ALIAS_PREFIX}{tool['name']}"
aliased["name"] = alias
rewritten.append(aliased)
else:
rewritten.append(tool)
return rewritten
return rewritten, alias_map
def _xai_prefers_native_web_search() -> bool:
@@ -407,6 +459,14 @@ class ResponsesApiTransport(ProviderTransport):
# attribute default; mutated on the instance, not the class.
_last_issuer_kind: Optional[str] = None
# Wire-alias provenance of the most recent build_kwargs call:
# ``{alias_sent_on_wire: original_tool_name}``. ``None`` means "no
# request built on this instance" (normalize-only call sites), in which
# case normalize_response falls back to the static legacy map. An empty
# dict means the last request emitted no aliases, so no reverse rewrite
# is permitted (#95003 provenance contract).
_last_wire_aliases: Optional[Dict[str, str]] = None
@property
def api_mode(self) -> str:
return "codex_responses"
@@ -606,6 +666,12 @@ class ResponsesApiTransport(ProviderTransport):
# is honored, but rename the wire tool to
# ``hermes_web_search`` so Grok cannot hijack the name. The alias
# is mapped back to ``web_search`` in ``normalize_response``.
# Request-local alias provenance: every wire alias THIS request
# emits is recorded here and stashed on the transport, so the
# reverse rewrite in ``normalize_response`` applies only to aliases
# that were actually sent (never to a real tool that merely shares
# an alias-shaped name).
wire_aliases: Dict[str, str] = {}
if is_xai_responses and response_tools:
has_client_web_search = any(
isinstance(t, dict) and t.get("name") == "web_search"
@@ -621,12 +687,30 @@ class ResponsesApiTransport(ProviderTransport):
response_tools = filtered
else:
response_tools = _rename_client_web_search_for_xai(response_tools)
wire_aliases[_XAI_CLIENT_WEB_SEARCH_ALIAS] = "web_search"
# OpenCode Responses backends reserve web_search / search_files as
# function names (HTTP 400 "custom function name 'X' is reserved",
# #85589). Alias them on the wire; normalize_response maps them back.
if response_tools and _is_opencode_responses_backend(params):
response_tools = _rename_reserved_tools_for_opencode(response_tools)
response_tools, _oc_aliases = _alias_reserved_tools(
response_tools, _OPENCODE_RESERVED_TOOL_NAMES
)
wire_aliases.update(_oc_aliases)
# xAI reserves ``tool_search`` for its native server-side tool and
# rejects the client declaration outright (#95003). Alias it on the
# wire; normalize_response maps it back before dispatch.
if is_xai_responses and response_tools:
response_tools, _xai_aliases = _alias_reserved_tools(
response_tools, _XAI_RESERVED_TOOL_NAMES
)
wire_aliases.update(_xai_aliases)
# Stash for normalize_response (same request/response pairing model
# as ``_last_issuer_kind``). An empty dict is meaningful: it means
# this request emitted NO aliases, so no reverse rewrite may run.
self._last_wire_aliases = wire_aliases
# ``tools`` MUST be omitted entirely when there are no functions to
# expose: the openai SDK's ``responses.stream()`` / ``responses.parse()``
@@ -864,14 +948,22 @@ class ResponsesApiTransport(ProviderTransport):
if hasattr(tc, "response_item_id") and tc.response_item_id:
provider_data["response_item_id"] = tc.response_item_id
name = tc.function.name if hasattr(tc, "function") else getattr(tc, "name", "")
# Undo the xAI client-path wire alias so Hermes dispatches
# the real ``web_search`` tool (Firecrawl / etc.).
if name == _XAI_CLIENT_WEB_SEARCH_ALIAS:
name = "web_search"
# Undo the OpenCode reserved-name wire aliases the same way
# (hermes_web_search / hermes_search_files, #85589).
elif name in _RESERVED_ALIAS_TO_NAME:
name = _RESERVED_ALIAS_TO_NAME[name]
# Undo THIS request's wire aliases before Hermes dispatch.
# Request-local provenance: only aliases the paired
# build_kwargs call actually emitted are rewritten, so a
# legitimate tool that happens to be named
# ``hermes_tool_search`` etc. is dispatched as itself when
# no alias was sent. The static legacy map is used only for
# normalize-only call sites that never built a request on
# this transport instance.
alias_map = self._last_wire_aliases
if alias_map is None:
if name == _XAI_CLIENT_WEB_SEARCH_ALIAS:
name = "web_search"
elif name in _LEGACY_ALIAS_FALLBACK:
name = _LEGACY_ALIAS_FALLBACK[name]
elif name in alias_map:
name = alias_map[name]
tool_calls.append(ToolCall(
id=tc.id if hasattr(tc, "id") else (name or None),
name=name,
+6
View File
@@ -133,6 +133,12 @@ class NormalizedResponse:
pd = self.provider_data or {}
return pd.get("anthropic_content_blocks")
@property
def bedrock_content_blocks(self):
"""Verbatim, order-preserving Bedrock Converse content blocks."""
pd = self.provider_data or {}
return pd.get("bedrock_content_blocks")
@property
def codex_reasoning_items(self):
pd = self.provider_data or {}
+93 -4
View File
@@ -95,13 +95,39 @@ def _preflight_request_tokens(
"using generic transcript estimate",
exc_info=True,
)
if _agent_stale_thinking_on_wire(agent):
return estimate_request_tokens_rough(
messages,
system_prompt=system_prompt or "",
tools=tools,
)
return estimate_request_tokens_rough(
messages,
system_prompt=system_prompt or "",
tools=tools,
charge_stale_thinking=False,
)
def _agent_stale_thinking_on_wire(agent: Any) -> bool:
"""Whether the agent's active route replays stale thinking text (#84371).
Route facts unavailable (test doubles, partially-built agents) default to
``True`` — the conservative full charge.
"""
try:
from agent.message_sanitization import stale_thinking_reaches_wire
return stale_thinking_reaches_wire(
getattr(agent, "api_mode", "") or "",
getattr(agent, "provider", "") or "",
getattr(agent, "model", "") or "",
getattr(agent, "base_url", "") or "",
)
except Exception:
return True
def compose_user_api_content(
content: Any,
ext_prefetch_cache: str,
@@ -379,6 +405,23 @@ def compression_made_progress(
_compression_made_progress = compression_made_progress
class PreflightCompressionTimedOut(RuntimeError):
"""Raised when an oversized turn cannot safely finish preflight."""
def _fail_closed_after_preflight_timeout(agent, request_tokens: int) -> None:
"""Stop an oversized turn instead of sending its unchanged provider payload."""
from agent.conversation_compression import context_compression_timed_out
if not context_compression_timed_out(agent):
return
raise PreflightCompressionTimedOut(
"Context compression timed out before it could commit while the request "
f"was still approximately {request_tokens:,} tokens. The provider call "
"was not sent. Run /compress and wait for it to finish, then retry."
)
def _review_fork_first_request_pending(agent: Any) -> bool:
"""Whether a detached review fork has yet to send its first provider request.
@@ -514,6 +557,7 @@ def build_turn_context(
stream_callback,
persist_user_message: Optional[Any],
persist_user_timestamp: Optional[float] = None,
persist_user_platform_id: Optional[str] = None,
*,
persist_user_display_kind: Optional[str] = None,
persist_user_display_metadata: Optional[Dict[str, Any]] = None,
@@ -627,6 +671,7 @@ def build_turn_context(
agent._persist_user_message_idx = None
agent._persist_user_message_override = persist_user_message
agent._persist_user_message_timestamp = persist_user_timestamp
agent._persist_user_message_platform_id = persist_user_platform_id
# Generate unique task_id if not provided to isolate VMs between tasks.
effective_task_id = task_id or str(uuid.uuid4())
agent._current_task_id = effective_task_id
@@ -764,6 +809,13 @@ def build_turn_context(
if persist_user_display_metadata:
user_msg["display_metadata"] = persist_user_display_metadata
# Stamp the platform-side message id (e.g. the Discord/Telegram message id)
# as metadata on the user turn so it survives the early crash-resilience
# persist below (the turn-start flush). Load-bearing for restart
# drain-window recovery: a recovery pass dedups via
# ``has_platform_message_id`` against this row.
if persist_user_platform_id is not None:
user_msg["platform_message_id"] = persist_user_platform_id
append_message(messages, user_msg)
current_turn_user_idx = len(messages) - 1
agent._persist_user_message_idx = current_turn_user_idx
@@ -1014,10 +1066,18 @@ def build_turn_context(
)
if not _preflight_deferred:
_last = _compressor.last_prompt_tokens
# Do NOT overwrite the -1 sentinel (#36718).
if _last >= 0 and _preflight_tokens > _last:
_compressor.last_prompt_tokens = _preflight_tokens
# Display-only seed (see
# ContextCompressor.maybe_seed_preflight_display_tokens): a real
# provider reading always wins over the rough estimate, and the
# -1 post-compression sentinel (#36718) stays protected. On
# usage-less responses the seed also feeds the tool-loop
# compression gate — the one live path where an inflated seed
# could push compression below the user threshold.
_maybe_seed = getattr(
_compressor, "maybe_seed_preflight_display_tokens", None
)
if callable(_maybe_seed):
_maybe_seed(_preflight_tokens)
_compression_cooldown = getattr(
_compressor,
@@ -1068,6 +1128,34 @@ def build_turn_context(
_compress_block_reason = _info(_preflight_tokens)[1]
except Exception:
_compress_block_reason = None
if _should_compress_now:
# Managed local runtime: growing the window beats compressing —
# the ladder's design order (same seam as the conversation
# loop's pre-API gate; see _maybe_grow_local_window there).
try:
from agent.conversation_loop import _maybe_grow_local_window
_grown = _maybe_grow_local_window(
agent, _compressor, _preflight_tokens
)
except Exception:
_grown = None
if _grown:
_compressor.update_model(
agent.model,
_grown,
base_url=getattr(agent, "base_url", "") or "",
api_key=getattr(agent, "api_key", "") or "",
provider=getattr(agent, "provider", "") or "",
api_mode=getattr(agent, "api_mode", "") or "",
)
agent._buffer_status(
f"📈 Context window grown to {_grown // 1024}K "
f"(local model; conversation continues uncompressed)"
)
_should_compress_now = _compressor.should_compress(
_preflight_tokens
)
if _should_compress_now:
_preflight_compressed = True
# Compression is actually running (block cleared / was never
@@ -1143,6 +1231,7 @@ def build_turn_context(
if not _compression_made_progress(
_orig_len, len(messages), _orig_tokens, _preflight_tokens
):
_fail_closed_after_preflight_timeout(agent, _preflight_tokens)
_preflight_compression_blocked = True
break # Cannot compress further: neither rows nor tokens moved
conversation_history = conversation_history_after_compression(
+1 -1
View File
@@ -752,7 +752,7 @@ def finalize_turn(
"health (`hermes doctor`), then send your message again"
)
# Machine-readable cause for the gateway/desktop: exactly
# 'session_persistence_failed:<locked|compression|turn_lease|corrupt|disk|unknown>'.
# 'session_persistence_failed:<locked|compression|turn_lease|corrupt|replaced|disk|unknown>'.
# Never clobber a failure_reason another path already stamped.
if "failure_reason" not in result:
_cause = getattr(agent, "_last_persistence_error_cause", None)
+343
View File
@@ -0,0 +1,343 @@
"""Turn liveness watchdog (#95548): force-abort turns that stall silently.
A conversation turn can stall mid-flight (observed in #95548 between
"model returned tool_calls" and tool execution, after a slow model
response + desktop WS disconnect) with no error logged, no further
progress, and the durable session turn lease kept renewing — so nothing
ever force-aborts the turn and the session stays stuck until the process
is killed.
This module owns the watchdog policy end to end:
* configuration — :func:`resolve_turn_liveness_settings` reads the
``agent.turn_liveness`` section of config.yaml and validates it;
* state machine — :class:`TurnLivenessWatchdog` samples the agent's
activity clock and drives the stall decision;
* thread mechanics — the polling loop, stop handling, and the stall
commit point.
``AIAgent.run_conversation`` (``run_agent.py``) keeps only the smallest
integration seam: it resolves the settings, supplies the commit and
deactivate callbacks that own turn-lease state, and starts the thread
next to the durable lease refresher.
Config surface (config.yaml)::
agent:
turn_liveness:
timeout_s: 600.0 # idle bound; <= 0 disables the watchdog
poll_s: 15.0 # sampling interval (seconds)
Both values are validated. A non-numeric typo, ``NaN``, or ``Inf`` logs a
warning and falls back to the documented default — it never crashes
startup, and a bogus value can never silently disable the watchdog
(``NaN``) or freeze the watcher thread (``Inf`` poll).
Race safety (#95663 review): the watchdog samples the activity clock and
binds the abort decision to the observed ``(generation, timestamp)`` pair.
The commit callback revalidates that pair under the *same* lock
``AIAgent._touch_activity`` uses to stamp the clock, so a turn that
resumed while the stall was being logged/emitted is never hard-cancelled
— it continues and its lease keeps renewing. The revalidated generation
is carried into ``AIAgent.interrupt`` (``require_generation``) as a
cancellation claim consumed at the final mutation edge: ``interrupt``
reserves the claim under the activity lock, ``_touch_activity``
invalidates the reservation the instant real progress lands, and the
claim survives every blocking boundary, including the compression
commit fence. Claim consumption and the first observable interrupt
state publish in ONE activity-lock critical section, so a turn that
resumes can only ever interleave before that section (the reservation
is invalidated and the abort declines) or after it (the interrupt
already committed under the lock) — never between "claim consumed" and
"state published". A turn that resumes anywhere in the window is never
hard-cancelled, and an exceptional interrupt path declines the abort
fail-closed instead of mutating interrupt state.
"""
from __future__ import annotations
import logging
import math
import threading
import time
from typing import Any, Callable, Dict, NamedTuple, Optional, Tuple
logger = logging.getLogger(__name__)
DEFAULT_TURN_LIVENESS_TIMEOUT_S = 600.0
DEFAULT_TURN_LIVENESS_POLL_S = 15.0
MIN_TURN_LIVENESS_POLL_S = 0.01
_CONFIG_TIMEOUT_KEY = "agent.turn_liveness.timeout_s"
_CONFIG_POLL_KEY = "agent.turn_liveness.poll_s"
class ActivitySnapshot(NamedTuple):
"""One observation of the activity clock, bound to an abort decision.
``generation`` + ``activity_ts`` uniquely identify the observed stamp.
The commit callback must revalidate this pair under the lock shared
with ``AIAgent._touch_activity``; if it no longer matches, the
observation is stale and the abort must be declined.
"""
generation: int
activity_ts: Optional[float]
idle_seconds: float
def _warn_invalid_value(key: str, raw: Any, default: float) -> None:
logger.warning(
"Invalid %s in config.yaml: %r — falling back to default %.1f.",
key,
raw,
default,
)
def _resolve_finite_seconds(raw: Any, *, default: float, key: str) -> float:
"""Coerce one duration knob, rejecting typos, NaN and Inf.
A non-numeric value must not raise into durable-turn startup, and a
non-finite value must not silently change behavior (``NaN`` would
disable the timeout via the ``> 0`` comparison; ``Inf`` would freeze
the poll loop in ``Event.wait``).
"""
try:
value = float(raw)
except (TypeError, ValueError):
_warn_invalid_value(key, raw, default)
return default
if not math.isfinite(value):
_warn_invalid_value(key, raw, default)
return default
return value
def resolve_turn_liveness_settings(
config: Optional[Dict[str, Any]] = None,
) -> Tuple[Optional[float], float]:
"""Resolve ``(timeout_s, poll_s)`` from the ``agent.turn_liveness`` section.
Precedence: explicit config.yaml value wins over the default; any
invalid value (typo, NaN, Inf, non-positive poll) falls back to the
default with a warning. ``timeout_s <= 0`` is the documented opt-out
and yields ``(None, poll_s)`` — the caller then never arms the
watchdog. The resolver never raises.
"""
section: Dict[str, Any] = {}
if isinstance(config, dict):
agent_cfg = config.get("agent")
if isinstance(agent_cfg, dict):
raw_section = agent_cfg.get("turn_liveness")
if isinstance(raw_section, dict):
section = raw_section
elif raw_section is not None:
_warn_invalid_value(
"agent.turn_liveness", raw_section, DEFAULT_TURN_LIVENESS_TIMEOUT_S
)
timeout_s = _resolve_finite_seconds(
section.get("timeout_s", DEFAULT_TURN_LIVENESS_TIMEOUT_S),
default=DEFAULT_TURN_LIVENESS_TIMEOUT_S,
key=_CONFIG_TIMEOUT_KEY,
)
poll_s = _resolve_finite_seconds(
section.get("poll_s", DEFAULT_TURN_LIVENESS_POLL_S),
default=DEFAULT_TURN_LIVENESS_POLL_S,
key=_CONFIG_POLL_KEY,
)
if poll_s <= 0:
_warn_invalid_value(_CONFIG_POLL_KEY, poll_s, DEFAULT_TURN_LIVENESS_POLL_S)
poll_s = DEFAULT_TURN_LIVENESS_POLL_S
# <= 0 is the documented opt-out, not an error.
if timeout_s <= 0:
timeout_s = None
return timeout_s, poll_s
class TurnLivenessWatchdog:
"""Sampled-idle watchdog thread bound to one conversation turn.
``run_agent.py`` owns the turn-lease state (stop event, turn-active
flag, interrupt plumbing); this class only reads the activity clock
and calls back at the stall commit point. All synchronization with
``AIAgent._touch_activity`` goes through ``activity_lock``, which
must be the SAME lock the agent stamps its activity clock with.
"""
def __init__(
self,
agent: Any,
*,
session_id: str,
timeout_s: float,
poll_s: float,
stop_event: threading.Event,
activity_lock: threading.Lock,
is_turn_active: Callable[[], bool],
commit_abort: Callable[[ActivitySnapshot, str], bool],
deactivate_turn: Callable[[], None],
) -> None:
self._agent = agent
self._session_id = session_id
self._timeout_s = float(timeout_s)
self._poll_s = max(MIN_TURN_LIVENESS_POLL_S, float(poll_s))
self._stop_event = stop_event
self._activity_lock = activity_lock
self._is_turn_active = is_turn_active
self._commit_abort = commit_abort
self._deactivate_turn = deactivate_turn
def make_thread(self) -> threading.Thread:
"""Build the (not yet started) watcher thread.
``run_agent.py`` creates the watchdog before the turn begins but
starts the thread at turn entry, right after the turn-active flag
and the activity clock are stamped.
"""
return threading.Thread(
target=self._watch,
name="turn-liveness-watchdog",
daemon=True,
)
def start(self) -> threading.Thread:
"""Spawn the watcher thread and return it (already running)."""
thread = self.make_thread()
thread.start()
return thread
def _watch(self) -> None:
while not self._stop_event.wait(self._poll_s):
snapshot = self._sample()
if snapshot is None:
# Turn is no longer active; nothing to watch.
return
if snapshot.idle_seconds < self._timeout_s:
continue
# Pre-commit surface is OBSERVATIONAL only: it reports the
# stall and that a recovery attempt is beginning. It must not
# claim the abort or the lease withdrawal has committed — the
# next operation can still veto the outcome. The definitive
# aborted/lease-stopped settlement is published by
# _surface_committed_abort only after _commit_abort succeeds
# and the turn is deactivated (#95663 review).
self._surface_stall(snapshot)
# Commit point: bind the abort to the sampled generation/ts
# and revalidate under the lock shared with `_touch_activity`.
# If progress resumed while the stall was being surfaced, the
# turn continues and this loop resumes sampling — the lease
# keeps renewing. The commit also carries the revalidated
# generation into the interrupt path, which reserves it as a
# claim, survives every blocking boundary (compression
# fence), and consumes it at the final mutation edge — progress
# landing anywhere in that window declines the abort.
if not self._commit_abort(snapshot, self._abort_message(snapshot)):
continue
# Stop renewing the durable lease: a wedge the hard interrupt
# cannot unwind must not keep the lease alive forever (the
# issue's "lease keeps renewing" masking). The TTL expiry then
# lets stale-turn cleanup reclaim the row.
self._deactivate_turn()
self._surface_committed_abort(snapshot)
return
def _sample(self) -> Optional[ActivitySnapshot]:
with self._activity_lock:
if not self._is_turn_active():
return None
generation = getattr(
self._agent, "_turn_liveness_activity_generation", 0
)
activity_ts = getattr(self._agent, "_last_activity_ts", None)
now = time.time()
if activity_ts is None:
idle_seconds = 0.0
else:
idle_seconds = max(0.0, now - activity_ts)
return ActivitySnapshot(
generation=generation,
activity_ts=activity_ts,
idle_seconds=idle_seconds,
)
def _abort_message(self, snapshot: ActivitySnapshot) -> str:
return (
f"Turn made no progress for {int(snapshot.idle_seconds)}s; "
"aborting to release the session."
)
def _surface_stall(self, snapshot: ActivitySnapshot) -> None:
"""Observationally surface the stall: log it loudly and emit a
UI-visible warning that a recovery attempt is beginning.
Deliberately does NOT claim the abort or the lease withdrawal has
committed: the next operation (``_commit_abort``) can still veto
the outcome when the turn resumed while this surface window was
open. The definitive aborted/lease-stopped settlement is
published by :meth:`_surface_committed_abort` only after the
abort wins and the turn is deactivated.
Rate-limited: a turn whose aborts keep declining (resumed
activity, exceptional interrupt path) must not re-log and
re-warn every poll interval — the first surface carries the
signal, repeats are suppressed until activity actually moves
again (a new generation re-arms the surface).
"""
generation = snapshot.generation
if getattr(self, "_last_surfaced_generation", None) == generation:
return
self._last_surfaced_generation = generation
session_id = getattr(self._agent, "session_id", None) or self._session_id
last_desc = getattr(self._agent, "_last_activity_desc", None)
logger.error(
"Turn liveness watchdog fired for session %s: "
"no progress for %.1fs (last activity: %r). "
"Attempting recovery: force-interrupting the turn and "
"stopping lease renewal if it cannot resume (#95548).",
session_id,
snapshot.idle_seconds,
last_desc,
)
emit_warning = getattr(self._agent, "_emit_warning", None)
if not callable(emit_warning):
return
try:
emit_warning(
"⚠️ This turn stopped making progress "
f"({int(snapshot.idle_seconds)}s without activity); "
"attempting recovery so the session can continue."
)
except Exception:
logger.debug("Failed to emit turn liveness warning", exc_info=True)
def _surface_committed_abort(self, snapshot: ActivitySnapshot) -> None:
"""Publish the definitive settlement AFTER the abort has authority.
Runs only once ``_commit_abort`` succeeded (the interrupt was
published) and the turn lease was deactivated: the turn IS
force-aborted and lease renewal IS stopped, so stating that is
now true. Separated from the pre-commit surface so a declined
abort never reports a committed outcome (#95663 review).
"""
session_id = getattr(self._agent, "session_id", None) or self._session_id
logger.error(
"Turn liveness watchdog aborted turn for session %s: "
"no progress for %.1fs; turn interrupted and lease renewal "
"stopped (#95548).",
session_id,
snapshot.idle_seconds,
)
emit_warning = getattr(self._agent, "_emit_warning", None)
if not callable(emit_warning):
return
try:
emit_warning(
"⚠️ Turn aborted by the liveness watchdog "
f"({int(snapshot.idle_seconds)}s without activity); "
"lease renewal stopped so the session can be reclaimed. "
"You can retry your message."
)
except Exception:
logger.debug("Failed to emit committed-abort warning", exc_info=True)
+1 -1
View File
@@ -14,7 +14,7 @@ Providers live in ``<repo>/plugins/web/<name>/`` (built-in, auto-loaded as
This ABC is the SINGLE plugin-facing surface for web providers — every
provider in the tree (brave-free, ddgs, searxng, exa, parallel, tavily,
firecrawl) implements it. The legacy in-tree ``tools.web_providers.base``
keenable, firecrawl) implements it. The legacy in-tree ``tools.web_providers.base``
ABCs were deleted in PR #25182 along with the per-vendor inline helpers
in ``tools/web_tools.py``; the response-shape contract documented below
is preserved bit-for-bit so the tool wrapper does not have to translate.
+1 -2
View File
@@ -168,7 +168,7 @@ _LEGACY_PREFERENCE = (
# Keyless free-tier walk — strictly LAST-resort, tried only after the
# availability-filtered legacy walk finds nothing (i.e. the user has zero
# web credentials and no importable ddgs). All five vendors expose public
# web credentials and no importable ddgs). Ring vendors expose public
# anonymous free tiers (see plugins/web/keyless_mcp.py). Unpinned keyless
# traffic round-robins across the ring per request (the ring cursor lives
# in keyless_mcp; an explicit `hermes tools` pick bypasses this walk
@@ -177,7 +177,6 @@ _LEGACY_PREFERENCE = (
_KEYLESS_PREFERENCE = (
"exa",
"parallel",
"tavily",
"firecrawl",
"keenable",
)
@@ -20,8 +20,13 @@ import { expect, test } from './test'
// on every bot switch, because nothing records a close (the plugin keeps no
// closed set; core's tile bucket only forgets). Now a bot whose workspace
// already holds tabs comes back to the one the user left; the forever-chat is
// re-opened only by the explicit asks (row menu "Open Bot Chat", Bots home
// "Open chat").
// re-opened only by the explicit asks (row menu "Open Bot Chat").
//
// UI note (post design-system rework): the canonical Bot Chat opens INTO the
// main workspace pane (`data-tree-tab="workspace"`), and a lone uncloseable
// workspace pane renders chromeless — its "Bot Chat" tab only exists once a
// second pane (e.g. a ⌘/Ctrl+T thread tile) shares the main zone. Assertions
// about the lone open therefore read the transcript, not a tab.
type Page = MockBackendFixture['page']
@@ -76,7 +81,9 @@ async function snap(page: Page, name: string): Promise<void> {
}
}
/** The session tabs on the main strip (the Bots home tab may sit beside them). */
/** The session tabs on the main strip (the Bot Chat workspace tab may sit
* beside them). The strip itself auto-hides when the workspace pane is the
* only pane in the zone, so an empty result also covers "no strip at all". */
const mainTabs = (page: Page) =>
page.evaluate(() =>
[...document.querySelectorAll<HTMLElement>('[data-zone-tabstrip="grp-main"] [data-tree-tab]')]
@@ -89,7 +96,7 @@ const mainTabs = (page: Page) =>
* row (the plugin's canonical forever-chat, found by exact title) — keeps
* in-app creation and the intro turn it fires out of a scenario that is
* about the row click. With the row present, the click takes the open-as-
* tab path; without it, it would mint the chat into the workspace pane. */
* workspace path; without it, it would mint the chat into the pane. */
async function seedBot(hermesHome: string, mockUrl: string, name: string): Promise<void> {
const dir = path.join(hermesHome, 'profiles', name)
fs.mkdirSync(dir, { recursive: true })
@@ -146,37 +153,47 @@ test('a bot row click returns to the open thread and does not re-open a closed B
await expect(alphaRow).toBeVisible({ timeout: 30_000 })
await expect(betaRow).toBeVisible({ timeout: 30_000 })
const botChatTab = page.getByRole('tab', { name: /Bot Chat/ }).filter({ visible: true })
// The seeded forever-chat's first turn — visible only while the Bot Chat
// transcript is on screen. This is how a chromeless lone open is observed.
const seededTurn = page.getByText('Hello alpha', { exact: true }).filter({ visible: true })
// The first click on a bot with nothing open lands on its canonical chat.
// It fills the lone main workspace pane, which renders without a tab strip.
await openUntil(
() => alphaRow.click(),
() => expect(botChatTab.first()).toBeVisible({ timeout: 45_000 })
() => expect(seededTurn.first()).toBeVisible({ timeout: 45_000 })
)
await settle(page, 15_000)
await snap(page, '01-first-click-opens-bot-chat')
// Close it, then start a fresh thread for Alpha (⌘/Ctrl+T — the strip's
// "+" leaves with the zone's last tab).
await botChatTab.first().hover()
await botChatTab.first().getByRole('button', { name: 'Close' }).click({ force: true })
await expect(botChatTab).toHaveCount(0)
// Start a fresh thread for Alpha (⌘/Ctrl+T). The thread tile joins the main
// zone beside the Bot Chat workspace pane, which mounts the tab strip — the
// "Bot Chat" tab exists now, and the close affordance with it.
await page.keyboard.press('Control+t')
await expect(botChatTab.first()).toBeVisible({ timeout: 15_000 })
await expect.poll(() => mainTabs(page), { timeout: 15_000 }).toHaveLength(1)
const composer = page.locator('[data-slot="composer-root"] [contenteditable="true"]').filter({ visible: true }).first()
await expect(composer).toBeVisible({ timeout: 15_000 })
await composer.click()
await composer.fill('hello alpha thread')
await page.keyboard.press('Enter')
await expect(page.getByText('hello alpha thread').filter({ visible: true }).first()).toBeVisible({ timeout: 15_000 })
// The reply also becomes the tab's (clipped) title — match the visible copy.
await expect(page.getByText(MOCK_REPLY).filter({ visible: true }).first()).toBeVisible({ timeout: 60_000 })
await snap(page, '02-closed-bot-chat-new-thread')
await snap(page, '02-new-thread-beside-bot-chat')
const threadTabs = await mainTabs(page)
expect(threadTabs).toHaveLength(1)
const [threadTab] = threadTabs
expect(threadTab).toMatch(/^session-tile:/)
// Close the Bot Chat. Its transcript leaves the screen; the thread stays.
await botChatTab.first().hover()
await botChatTab.first().getByRole('button', { name: 'Close' }).click({ force: true })
await expect(botChatTab).toHaveCount(0)
await expect(seededTurn).toHaveCount(0)
await snap(page, '03-bot-chat-closed-thread-stays')
// Switch to Beta: Alpha's thread leaves the strip (scoped away, not closed).
await betaRow.click()
await expect(page.locator(`[data-zone-tabstrip="grp-main"] [data-tree-tab="${threadTab}"]`)).toHaveCount(0, {
@@ -184,25 +201,27 @@ test('a bot row click returns to the open thread and does not re-open a closed B
})
await settle(page)
// Back to Alpha: the thread is fronted, and the closed Bot Chat STAYS closed.
// Back to Alpha: the workspace comes back to what the user left, and the
// closed Bot Chat STAYS closed. The regression this pins re-opened the
// canonical chat beside the thread on every switch — two panes in the main
// zone, which mounts the tab strip and puts the "Bot Chat" tab back on
// screen. Its absence (with the transcript present, so the click landed) is
// the observable "stays closed".
await alphaRow.click()
const threadTabLocator = page.locator(`[data-zone-tabstrip="grp-main"] [data-tree-tab="${threadTab}"]`)
await expect(threadTabLocator).toBeVisible({ timeout: 30_000 })
await expect(threadTabLocator).toHaveAttribute('aria-selected', 'true')
await expect(page.getByText(MOCK_REPLY).filter({ visible: true }).first()).toBeVisible({ timeout: 30_000 })
await page.waitForTimeout(3000)
await expect(botChatTab).toHaveCount(0)
expect(await mainTabs(page)).toEqual([threadTab])
await snap(page, '03-back-to-alpha-bot-chat-stays-closed')
await snap(page, '04-back-to-alpha-bot-chat-stays-closed')
// The explicit ask still opens the forever-chat, beside the thread.
// The explicit ask still opens the forever-chat: its seeded first turn is
// back on screen. (As the surviving main-workspace pane it may render
// chromeless, so the transcript — not a tab — is the assertion.)
await openUntil(
async () => {
await alphaRow.click({ button: 'right' })
await page.getByRole('menuitem', { name: 'Open Bot Chat' }).click()
},
() => expect(botChatTab.first()).toBeVisible({ timeout: 45_000 })
() => expect(seededTurn.first()).toBeVisible({ timeout: 45_000 })
)
expect(await mainTabs(page)).toHaveLength(2)
expect(await mainTabs(page)).toContain(threadTab)
await snap(page, '04-explicit-open-bot-chat')
await snap(page, '05-explicit-open-bot-chat')
})
+6 -5
View File
@@ -106,7 +106,10 @@ test.describe('chat interaction with mock backend', () => {
await composer.click()
await composer.type('please answer tersely')
await expect(primary).toHaveAttribute('aria-label', /Steer/)
// Since "running is not busy" (3bc52fb9df) the primary keeps the Send
// affordance mid-turn — steer is routed through the submit engine, not a
// separate labeled button. Queue remains the explicit secondary action.
await expect(primary).toHaveAttribute('aria-label', 'Send')
await expect(dictation).toBeVisible()
await expect(speakReplies).toBeVisible()
await expect(queue).toBeVisible()
@@ -119,11 +122,9 @@ test.describe('chat interaction with mock backend', () => {
)
expect(controlLabels.indexOf('Voice dictation')).toBeLessThan(speakRepliesIndex)
expect(speakRepliesIndex).toBeLessThan(controlLabels.indexOf('Queue message'))
expect(controlLabels.indexOf('Queue message')).toBeLessThan(
controlLabels.findIndex(label => label?.startsWith('Steer'))
)
expect(controlLabels.indexOf('Queue message')).toBeLessThan(controlLabels.indexOf('Send'))
await page.screenshot({ path: testInfo.outputPath('busy-composer-steer.png') })
await expect(primary.locator('svg.tabler-icon-steering-wheel')).toBeVisible()
await expect(primary.locator('.codicon-arrow-up')).toBeVisible()
await queue.click()
await expect(primary).toHaveAttribute('aria-label', 'Stop')
@@ -45,7 +45,9 @@ async function steer(page: Page, text: string): Promise<void> {
await composer.waitFor({ state: 'visible', timeout: 15_000 })
await composer.click()
await composer.type(text, { delay: 5 })
await expect(primary).toHaveAttribute('aria-label', /Steer/)
// Since "running is not busy" (3bc52fb9df) the primary keeps the Send label
// mid-turn; the submit engine still routes a text payload to steer.
await expect(primary).toHaveAttribute('aria-label', 'Send')
await primary.click()
}
@@ -209,17 +211,36 @@ test.describe('correction session switch', () => {
// Reproduce the observed race: switch to another persisted session while
// the foreground tool is live, then return before its redirect settles.
await openSidebarSession(page, MOCK_REPLY, OTHER_SESSION_PROMPT)
// Sidebar rows title by the session's first user prompt (auto-title is
// disabled in the e2e fixture config).
await openSidebarSession(page, OTHER_SESSION_PROMPT, OTHER_SESSION_PROMPT)
await reopenOriginalSession(page)
await page.waitForTimeout(500)
// The warm resume first paints the persisted history and then reconciles
// the live turn (including a steer whose persistence may lag on a loaded
// runner) back in. Poll to the converged order instead of sampling one
// arbitrary mid-reconcile frame; the duplicate checks then pin the
// regression (the prompt/correction must appear exactly once).
await expect
.poll(async () => relevantOrder(await transcriptTextOrder(page)), {
message: 'correction should stay in place after the warm resume',
timeout: 30_000,
})
.toEqual(orderBeforeSwitch)
await page.screenshot({ path: testInfo.outputPath('correction-after-warm-resume.png') })
expect(relevantOrder(await transcriptTextOrder(page))).toEqual(orderBeforeSwitch)
expect(await textNodeOccurrences(page, ORIGINAL_PROMPT)).toBe(1)
expect(await textNodeOccurrences(page, CORRECTION)).toBe(1)
await waitForTranscriptText(page, CORRECTED_REPLY)
expect(steerTurnOrder(await transcriptMessageOrder(page))).toEqual([ORIGINAL_PROMPT, CORRECTION, CORRECTED_REPLY])
// The post-turn stored-history reconcile can momentarily repaint from a
// snapshot in which the steer's user row hasn't been folded back in yet —
// poll to the converged order instead of sampling one frame.
await expect
.poll(async () => steerTurnOrder(await transcriptMessageOrder(page)), {
message: 'steered turn should settle as prompt → correction → corrected reply',
timeout: 30_000,
})
.toEqual([ORIGINAL_PROMPT, CORRECTION, CORRECTED_REPLY])
})
test('keeps an inference-time correction visible through a warm session switch', async ({}, testInfo: TestInfo) => {
@@ -236,7 +257,7 @@ test.describe('correction session switch', () => {
await send(page, INFERENCE_CORRECTION)
await waitForTranscriptText(page, INFERENCE_CORRECTION)
await openSidebarSession(page, MOCK_REPLY, OTHER_SESSION_PROMPT)
await openSidebarSession(page, OTHER_SESSION_PROMPT, OTHER_SESSION_PROMPT)
await reopenInferenceSession(page)
expect(await textNodeOccurrences(page, INFERENCE_PROMPT)).toBe(1)
+100
View File
@@ -0,0 +1,100 @@
/**
* Locating the dev Electron binary for the e2e fixtures.
*
* Kept in its own module so the resolution rules can be unit-tested without
* importing the Playwright runner (fixtures.ts pulls in `_electron`, the mock
* server and the error-banner guard).
*
* Three rules the previous single-path probe got wrong:
*
* 1. The binary is not always under the REPO ROOT. This is an npm workspaces
* repo, and npm only hoists a dependency to the root when nothing conflicts
* — otherwise `electron` installs into `apps/desktop/node_modules`. Both
* layouts are normal, so both have to be searched, nearest package first.
* 2. The binary is `electron.exe` on Windows. A bare `electron` never exists
* there, so the probe could only ever miss.
* 3. `which` is not a command on Windows. The PATH fallback spawned it
* unconditionally, so on Windows the fallback failed for the wrong reason
* and the error message blamed a missing `npm install`.
*/
import { spawnSync } from 'node:child_process'
import * as fs from 'node:fs'
import { createRequire } from 'node:module'
import * as path from 'node:path'
/** The dist file name: `electron.exe` on Windows, `electron` elsewhere. */
export function electronBinaryName(platform: NodeJS.Platform = process.platform): string {
return platform === 'win32' ? 'electron.exe' : 'electron'
}
/**
* Where an npm install can leave the binary, in probe order: nearest package
* first, so a workspace-local install wins over a stale hoisted one.
*/
export function electronDistCandidates(roots: string[], platform: NodeJS.Platform = process.platform): string[] {
return roots.map((root) => path.join(root, 'node_modules', 'electron', 'dist', electronBinaryName(platform)))
}
/** The PATH-lookup command for this platform. Windows has `where`, not `which`. */
export function pathLookupCommand(platform: NodeJS.Platform = process.platform): string {
return platform === 'win32' ? 'where' : 'which'
}
/**
* Ask the installed `electron` package where its own binary is.
*
* Its main export IS the absolute executable path, resolved from `path.txt`
* and honouring `ELECTRON_OVERRIDE_DIST_PATH`, so this covers layouts and
* overrides a hand-built path cannot know about. Returns null when the package
* is not resolvable from `from`, or when it does not hand back a path (the
* export is the Electron API object, not a path, when required from inside
* Electron itself).
*/
export function electronPackagePath(from: string): null | string {
try {
const resolved = createRequire(path.join(from, 'package.json'))('electron') as unknown
return typeof resolved === 'string' && resolved ? resolved : null
} catch {
return null
}
}
/**
* Resolve the Electron binary, or throw with the layouts that were searched.
*
* `roots` are searched in order; pass the desktop package before the repo root.
*/
export function resolveElectronBinary(roots: string[]): string {
for (const root of roots) {
const declared = electronPackagePath(root)
if (declared && fs.existsSync(declared)) {
return declared
}
}
for (const candidate of electronDistCandidates(roots)) {
if (fs.existsSync(candidate)) {
return candidate
}
}
// Nix devshells put `electron` on PATH with no node_modules copy at all.
const lookup = spawnSync(pathLookupCommand(), ['electron'], { encoding: 'utf8' })
if (lookup.status === 0 && lookup.stdout.trim()) {
// `where` reports every match, one per line; take the first.
const first = lookup.stdout.trim().split(/\r?\n/)[0].trim()
if (first) {
return first
}
}
throw new Error(
`Electron binary not found. Searched ${electronDistCandidates(roots).join(', ')} and PATH. ` +
'Run "npm install" from the repo root to install devDependencies.',
)
}
@@ -0,0 +1,54 @@
import * as path from 'node:path'
import { describe, expect, it } from 'vitest'
import { electronBinaryName, electronDistCandidates, pathLookupCommand } from './electron-binary'
// Platform is a parameter everywhere below rather than read from
// process.platform, so the Windows rules are pinned on the Linux CI runner too.
// Reading the real platform would leave every Windows-only rule untested.
describe('electronBinaryName', () => {
it('asks for electron.exe on Windows', () => {
expect(electronBinaryName('win32')).toBe('electron.exe')
})
it('asks for a bare electron everywhere else', () => {
expect(electronBinaryName('linux')).toBe('electron')
expect(electronBinaryName('darwin')).toBe('electron')
})
})
describe('electronDistCandidates', () => {
const desktop = path.join('repo', 'apps', 'desktop')
const repo = 'repo'
it('probes the workspace-local install before the hoisted one', () => {
// npm only hoists `electron` to the repo root when nothing conflicts, so
// apps/desktop/node_modules is an ordinary outcome of `npm install`, not a
// broken tree. Probing only the repo root is what makes the suite refuse to
// start with "run npm install" on a tree that has electron installed.
expect(electronDistCandidates([desktop, repo], 'linux')).toEqual([
path.join(desktop, 'node_modules', 'electron', 'dist', 'electron'),
path.join(repo, 'node_modules', 'electron', 'dist', 'electron'),
])
})
it('carries the platform binary name into every candidate', () => {
// A bare `electron` file never exists in a Windows dist, so a probe built
// from a hardcoded name cannot match there no matter which root it walks.
for (const candidate of electronDistCandidates([desktop, repo], 'win32')) {
expect(path.basename(candidate)).toBe('electron.exe')
}
})
})
describe('pathLookupCommand', () => {
it('uses where on Windows and which elsewhere', () => {
// `which` is not a command on Windows; spawning it unconditionally made the
// PATH fallback fail for a reason unrelated to whether electron is on PATH.
expect(pathLookupCommand('win32')).toBe('where')
expect(pathLookupCommand('linux')).toBe('which')
expect(pathLookupCommand('darwin')).toBe('which')
})
})
+32 -21
View File
@@ -20,13 +20,13 @@
* Prerequisite: `npm run build` must have been run so that `dist/` exists.
*/
import { spawnSync } from 'node:child_process'
import * as fs from 'node:fs'
import * as os from 'node:os'
import * as path from 'node:path'
import { _electron, type ElectronApplication, type Page } from '@playwright/test'
import { resolveElectronBinary } from './electron-binary'
import { startMockServer, type MockServerOptions } from './mock-server'
import { installErrorBannerGuard } from './test'
@@ -170,6 +170,29 @@ export function writeMockProviderConfig(
? `\ndisplay:\n${extraDisplayConfig}\n`
: ''
// Title generation rides the MAIN model since 87af576e60 (#83636), so every
// completed turn fires an extra background /v1/chat/completions at the mock.
// That request contains the whole conversation — trigger keywords included —
// which advances the mock's scripted-turn indices and trips hold-for-prompt
// matchers from a request no spec ever sent. Disable it by default (no e2e
// spec asserts on session titles); a test that passes its own `auxiliary:`
// section via extraConfig owns the whole section instead.
const autoTitleDefault = extraConfig?.includes('auxiliary:')
? ''
: 'auxiliary:\n title_generation:\n enabled: false\n'
// The scripted turns run REAL terminal commands, and anything the guard
// classifies as dangerous (e.g. the sidebar sentinel-wait loop) parks the
// turn behind a Run/Reject approval card. The default 'smart' mode then
// fires an aux LLM approval call at the SAME mock provider — consuming a
// scripted-turn index and never resolving — so the turn stalls until the
// spec times out (the CI failure mode for the sidebar-dot family). No e2e
// spec asserts on the approval flow, so run gate-free by default; a test
// that passes its own `approvals:` section via extraConfig owns it.
const approvalsDefault = extraConfig?.includes('approvals:')
? ''
: 'approvals:\n mode: "off"\n'
const config = `# Auto-generated by E2E test fixtures
model:
default: mock-model
@@ -183,7 +206,7 @@ ${modelContextLength ? ` context_length: ${modelContextLength}\n` : ''}provider
models:
mock-model: {}
context_length: 4096
${displaySection}${extraConfig ? `\n${extraConfig.trim()}\n` : ''}`
${autoTitleDefault}${approvalsDefault}${displaySection}${extraConfig ? `\n${extraConfig.trim()}\n` : ''}`
fs.writeFileSync(configPath, config, 'utf8')
}
@@ -281,30 +304,18 @@ function assertDistBuilt(): void {
/**
* Find the Electron binary. In the nix devshell, `electron` is on PATH.
* As a fallback, use the node_modules/.bin/electron from the desktop package.
* As a fallback, use the node_modules/electron install from either package.
*/
export function findElectron(): string {
// In dev mode, we use the `electron` binary directly (not the packaged app).
// The dev:electron script in package.json does exactly this: `electron .`
// after building. We replicate that here.
const localElectron = path.join(REPO_ROOT, 'node_modules', 'electron', 'dist', 'electron')
if (fs.existsSync(localElectron)) {
return localElectron
}
// Fall back to PATH
const result = spawnSync('which', ['electron'], {
encoding: 'utf8',
})
if (result.status === 0 && result.stdout.trim()) {
return result.stdout.trim()
}
throw new Error(
'Electron binary not found. Run "npm install" from the repo root to install devDependencies.',
)
//
// The desktop package is searched first: npm workspaces only hoist
// `electron` to the repo root when nothing conflicts, so a workspace-local
// install is just as ordinary an outcome as a hoisted one. The rules live in
// ./electron-binary so they can be unit-tested per platform.
return resolveElectronBinary([DESKTOP_ROOT, REPO_ROOT])
}
/**
+39 -8
View File
@@ -24,20 +24,46 @@ import { expect, type Page, test } from '@playwright/test'
import { type MockBackendFixture, setupMockBackend, waitForAppReady } from './fixtures'
const STRIP = '.glyph-spinner__strip'
/* Scope to a spinner that is actually RUNNING. Turns from earlier tests in
* this file leave parked spinners mounted (kept-alive panes, swap overlays
* hold them with data-paused='true'), and document.querySelector returns the
* FIRST strip in the DOM — a stale parked one once two turns have run. */
const STRIP = '.glyph-spinner:not([data-paused="true"]) .glyph-spinner__strip'
/** Prompt the mock server holds open so the spinner runs for the whole file. */
const SPINNER_PROMPT = 'E2E_GLYPH_SPINNER_HOLD'
/**
* Send a message so a turn is in flight — the composer status stack mounts a
* GlyphSpinner while the agent is working. Resolves once a frame strip is in
* the DOM.
* Get a RUNNING frame strip into the DOM deterministically.
*
* A turn is sent so the app is genuinely busy (the mock server holds the
* stream open), but which surface mounts a spinner mid-turn is app policy
* that has changed before and will again — the transcript, status stack and
* swap overlay all park/unmount theirs at different moments, which made this
* spec racy. The contract under test is the STYLESHEET (steps() animation,
* layer promotion, the data-paused and global pause gates), and that CSS is
* driven entirely by the `data-paused` attribute — the same attribute the
* parked assertions below already toggle. So: wait for any mounted spinner
* (the ChatSwapOverlay keeps one mounted, parked, after boot), then unpark it
* and assert against the running animation.
*/
async function mountSpinner(page: Page): Promise<void> {
if (await page.locator(STRIP).count()) {
return
}
const composer = page.locator('[contenteditable="true"]').first()
await composer.waitFor({ state: 'visible', timeout: 10_000 })
await composer.click()
await composer.type('hello from the glyph spinner spec', { delay: 10 })
await composer.type(SPINNER_PROMPT, { delay: 10 })
await page.keyboard.press('Enter')
await page.waitForSelector('.glyph-spinner__strip', { state: 'attached', timeout: 20_000 })
await page.evaluate(() => {
for (const el of document.querySelectorAll('.glyph-spinner[data-paused]')) {
el.removeAttribute('data-paused')
}
})
await page.waitForSelector(STRIP, { state: 'attached', timeout: 20_000 })
}
@@ -45,11 +71,14 @@ test.describe('GlyphSpinner (compositor animation)', () => {
let fixture: MockBackendFixture
test.beforeAll(async () => {
fixture = await setupMockBackend()
fixture = await setupMockBackend({
mockServer: { holdFirstStreamForPrompt: SPINNER_PROMPT },
})
await waitForAppReady(fixture)
})
test.afterAll(async () => {
fixture?.mock.releaseHeldStream()
await fixture?.cleanup()
})
@@ -93,8 +122,10 @@ test.describe('GlyphSpinner (compositor animation)', () => {
// multiple of the frame count — not the single-frame interval.
expect(observed.durationMs).toBeGreaterThan(0)
// Length-typed travel, never a percentage: `translateY(-100%)` would keep
// the animation off the compositor.
expect(observed.travel).toContain('calc(')
// the animation off the compositor. Chromium has serialized the resolved
// keyframe both as the authored `calc(...)` and as an absolute `...px`
// length depending on version — accept any length, reject percentages.
expect(observed.travel).toMatch(/calc\(|px\)/)
expect(observed.travel).not.toContain('%')
})
@@ -32,7 +32,7 @@ test.afterAll(async () => {
})
test('local bot replaces an open group main workspace', async () => {
test.setTimeout(180_000)
test.setTimeout(240_000)
const page = fixture!.page
await openBots(page)
@@ -60,11 +60,21 @@ test('local bot replaces an open group main workspace', async () => {
const programmer = page.getByRole('button', { name: /^Programmer\b/ }).filter({ visible: true }).first()
await programmer.click()
const botChatTab = page.getByRole('tab', { name: /Bot Chat Close/ }).filter({ visible: true })
await expect(botChatTab).toBeVisible({ timeout: 30_000 })
await expect(botChatTab).toHaveAttribute('aria-selected', 'true')
// The bot's canonical chat opens INTO the main workspace pane (post
// design-system rework); as the lone pane in the zone it renders chromeless
// — no "Bot Chat" tab exists until a second pane joins the strip. The
// handoff is observed by the group surfaces leaving and the bot's chat
// (here a fresh one: its empty-state splash asks for a first message)
// taking the main workspace. The first open also spawns the bot's own
// backend, so give the "Loading session" phase a real chance to clear.
await expect(page.getByText('Say something to get started.').filter({ visible: true })).toBeVisible({
timeout: 120_000
})
await expect(groupTab).toHaveCount(0)
await expect(groupComposer).toHaveCount(0)
await expect(page.getByText(/Waking up Programmer/i)).toHaveCount(0)
// No "Waking up…" assertion: the mock backend can keep a bot's wake notice
// around indefinitely (see bot-mode-closed-chat-stays-closed's settle()),
// so its presence no longer distinguishes a stranded handoff. The splash
// and composer above are the proof the bot's chat took the workspace.
await expect(page.locator('[data-slot="composer-root"] [contenteditable="true"]').filter({ visible: true }).first()).toBeVisible()
})
@@ -9,13 +9,11 @@
import * as fs from 'node:fs'
import * as path from 'node:path'
import { expect, test } from './test'
import {
type MockBackendFixture,
buildAppEnv,
createSandbox,
launchDesktop,
type MockBackendFixture,
waitForAppReady,
writeEnvFile,
writeMockProviderConfig,
@@ -27,6 +25,7 @@ import {
VERIFICATION_STOP_TRIGGER,
} from './mock-server'
import { RealSessionBuilder } from './real-session-builder'
import { expect, test } from './test'
const SESSION_TITLE = 'E2E Hidden History Messages'
const VISIBLE_USER_TEXT = 'E2E_VISIBLE_USER_HISTORY'
@@ -44,6 +43,7 @@ async function setupSeededMockBackend(): Promise<MockBackendFixture> {
)
writeEnvFile(sandbox.hermesHome)
const builder = await RealSessionBuilder.start(sandbox.hermesHome)
try {
await builder.createSession({
title: SESSION_TITLE,
@@ -83,6 +83,7 @@ test('resume hides real context-compaction handoffs', async ({}, testInfo) => {
.locator('[data-slot="sidebar"] button')
.filter({ hasText: SESSION_TITLE })
.first()
await sessionRow.click()
const transcript = page.locator('[data-slot="aui_thread-viewport"]')
@@ -110,8 +111,20 @@ test('live verify-on-stop continuations stay out of the transcript', async ({},
const mock = await startMockServer({ verificationWritePath: changedFile })
writeMockProviderConfig(sandbox.hermesHome, mock.url)
fs.appendFileSync(path.join(sandbox.hermesHome, 'config.yaml'), '\nagent:\n verify_on_stop: true\n', 'utf8')
// Auto session titling (feat f726090d48) fires an auxiliary title_generation
// LLM call whose user snippet CONTAINS the trigger keyword, so the mock's
// isVerificationStopTrigger matches it and the title call steals a scripted
// verify-on-stop turn (the transcript then ends on 'The code edit is
// complete.' instead of the exhausted-verifier final). Disable the
// model-backed title upgrade so script indices track real chat turns.
fs.appendFileSync(
path.join(sandbox.hermesHome, 'config.yaml'),
'\nauxiliary:\n title_generation:\n enabled: false\n',
'utf8',
)
writeEnvFile(sandbox.hermesHome)
const { app, page } = await launchDesktop(buildAppEnv(sandbox))
const fixture: MockBackendFixture = {
app,
page,
@@ -27,8 +27,8 @@ import { type MockServer, startMockServer } from './mock-server'
import { RealSessionBuilder } from './real-session-builder'
import { type ElectronApplication, expect, type Page, test } from './test'
// A seeded session has no generated title, so every label falls back to the
// session preview — the first 60 characters of the first user message.
// The builder-provided title now labels the sidebar row directly (seeded
// sessions no longer fall back to the first-user-message preview).
const SESSION_TITLE = 'E2E attached image session'
const CAPTION = 'E2E attached image must survive a relaunch'
const IMAGE_DIR = 'Application Support/e2e shots'
@@ -90,7 +90,7 @@ async function setupSeededDesktop(): Promise<SeededFixture> {
}
function sessionRow(page: Page) {
return page.locator('[data-slot="sidebar"] button').filter({ hasText: CAPTION }).first()
return page.locator('[data-slot="sidebar"] button').filter({ hasText: SESSION_TITLE }).first()
}
// Inactive tabs stay mounted under a data-pane-hidden ancestor. Match the
@@ -172,13 +172,15 @@ test.describe('attached image resume', () => {
fixture = await setupSeededDesktop()
await waitForAppReady(fixture, 120_000)
// The sidebar labels a session by its preview, so the caption has to lead
// the persisted turn — a leading directive reads as a truncated file path.
// The sidebar labels a seeded session by its title. Whatever the label
// source, an attachment directive must never leak into it as a file path.
const row = sessionRow(fixture.page)
await row.waitFor({ state: 'visible', timeout: 60_000 })
const label = (await row.textContent())?.trim() ?? ''
expect(label.startsWith(CAPTION), `sidebar label should open with the caption: ${label}`).toBe(true)
expect(label.startsWith(SESSION_TITLE), `sidebar label should open with the title: ${label}`).toBe(true)
expect(label, `sidebar label should not leak the image path: ${label}`).not.toContain(IMAGE_NAME)
expect(label, `sidebar label should not render the directive: ${label}`).not.toContain('@image:')
await openSeededSession(fixture.page)
await assertRendersThumbnail(fixture.page, 'first open')
+80 -20
View File
@@ -20,16 +20,24 @@
*
* display.interim_assistant_messages: true (default)
* → ALL interim texts AND the final text must be visible in the
* transcript.
* settled transcript.
*
* display.interim_assistant_messages: false
* → only the final text is visible (no message.interim events emitted,
* so all streamed interim text is replaced at message.complete).
* → no message.interim events are emitted, so no sealed interim bubbles
* are created while streaming. Since the post-turn stored-history
* reconcile (sessions.changed → reconcileActiveTranscript, commit
* 1a2b0ca8cb) converges the visible transcript to the persisted
* transcript — which has ALWAYS contained the mid-turn commentary as
* real assistant rows (that is what a resume shows, flag or no flag) —
* the settled DOM shows the whole turn as ONE assistant message
* containing commentary + final. The flag governs live sealing only.
* The test pins that converged single-message shape: every text
* appears exactly once, inside a single assistant message root.
*
* Prerequisite: `npm run build` must have been run so dist/ exists.
*/
import { expect, test, type Page } from '@playwright/test'
import { expect, type Page, test } from '@playwright/test'
import {
type MockBackendFixture,
@@ -40,6 +48,17 @@ import { INTERIM_TEXTS, restartMockServer } from './mock-server'
// ─── Helpers ──────────────────────────────────────────────────────────
/**
* Auto session titling (feat f726090d48, 2026-08-08) issues an auxiliary
* `title_generation` LLM call against the SAME provider as the chat turn.
* The mock server counts every completion request as a script turn, so the
* title call races the chat turn and steals a scripted interim turn (the
* stolen turn's text then never streams to the transcript). Disable the
* model-backed title upgrade — the instant derived title needs no LLM call —
* so the mock's script indices line up with real chat turns again.
*/
const DISABLE_AUTO_TITLE = 'auxiliary:\n title_generation:\n enabled: false'
/** Unique trigger keyword the mock server detects to switch to the script. */
const TRIGGER = 'E2E_INTERIM_TRIGGER'
@@ -72,7 +91,7 @@ async function sendInterimMessage(page: Page): Promise<void> {
)
// Give the renderer a moment to settle any final state updates
// (hydration, session refresh) before asserting.
// (hydration, stored-history reconcile, session refresh) before asserting.
await page.waitForTimeout(2000)
}
@@ -90,11 +109,13 @@ async function countTranscriptMessagesContaining(page: Page, text: string): Prom
return page.evaluate(
(search) => {
const viewport = document.querySelector('[data-slot="aui_thread-viewport"]')
if (!viewport) {
return 0
}
let count = 0
const walker = document.createTreeWalker(
viewport,
NodeFilter.SHOW_ELEMENT,
@@ -102,29 +123,46 @@ async function countTranscriptMessagesContaining(page: Page, text: string): Prom
acceptNode: (node) => {
const el = node as HTMLElement
const directText = el.textContent ?? ''
if (!directText.includes(search)) {
return NodeFilter.FILTER_SKIP
}
// Only count leaf-ish elements to avoid double-counting.
const hasChildWithText = Array.from(el.children).some(
(child) => (child.textContent ?? '').includes(search),
)
if (hasChildWithText) {
return NodeFilter.FILTER_SKIP
}
return NodeFilter.FILTER_ACCEPT
},
},
)
while (walker.nextNode()) {
count++
}
return count
},
text,
)
}
/** Count assistant message roots in the settled transcript. */
async function countAssistantMessageRoots(page: Page): Promise<number> {
return page.evaluate(() => {
const viewport = document.querySelector('[data-slot="aui_thread-viewport"]')
return viewport
? viewport.querySelectorAll('[data-slot="aui_assistant-message-root"]').length
: 0
})
}
// ─── Flag ON: interim_assistant_messages = true (default) ─────────────
test.describe('interim assistant messages — flag ON (default)', () => {
@@ -134,7 +172,7 @@ test.describe('interim assistant messages — flag ON (default)', () => {
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend()
fixture = await setupMockBackend({ extraConfig: DISABLE_AUTO_TITLE })
await waitForAppReady(fixture, 120_000)
})
@@ -147,8 +185,10 @@ test.describe('interim assistant messages — flag ON (default)', () => {
await sendInterimMessage(page)
// Every interim text (turns with visible text + tool calls) must be
// present in the transcript as its own sealed message — NOT wiped by
// message.complete.
// present in the settled transcript — NOT wiped by message.complete.
// (Live, each seals as its own bubble; the post-turn stored-history
// reconcile then converges the turn into one assistant message that
// still carries all of them.)
for (const interimText of INTERIM_TEXTS.interims) {
await expect
.poll(
@@ -165,6 +205,13 @@ test.describe('interim assistant messages — flag ON (default)', () => {
{ timeout: 15_000, message: 'final text should be visible' },
)
.toBeGreaterThanOrEqual(1)
// No duplicates: the reconcile must CONVERGE (replace the sealed live
// bubbles), never render a stored copy alongside a live one.
for (const text of [...INTERIM_TEXTS.interims, INTERIM_TEXTS.finalText]) {
const count = await countTranscriptMessagesContaining(page, text)
expect(count, `"${text}" must not be duplicated after reconcile`).toBe(1)
}
})
})
@@ -179,6 +226,7 @@ test.describe('interim assistant messages — flag OFF', () => {
restartMockServer()
fixture = await setupMockBackend({
extraDisplayConfig: ' interim_assistant_messages: false',
extraConfig: DISABLE_AUTO_TITLE,
})
await waitForAppReady(fixture, 120_000)
})
@@ -187,7 +235,7 @@ test.describe('interim assistant messages — flag OFF', () => {
await fixture?.cleanup()
})
test('only the final response is visible; all interim texts are wiped', async () => {
test('settled transcript converges to stored history as a single turn message', async () => {
const page = fixture.page
await sendInterimMessage(page)
@@ -199,17 +247,29 @@ test.describe('interim assistant messages — flag OFF', () => {
)
.toBeGreaterThanOrEqual(1)
// NONE of the interim texts should be visible — with the flag off,
// the tui_gateway never installs interim_assistant_callback, so no
// message.interim events are emitted. All streamed interim text is
// accumulated into the streaming bubble and replaced by
// message.complete.
for (const interimText of INTERIM_TEXTS.interims) {
const count = await countTranscriptMessagesContaining(page, interimText)
expect(
count,
`interim text "${interimText}" should NOT be visible when flag is off`,
).toBe(0)
// With the flag off, the tui_gateway never installs
// interim_assistant_callback, so no message.interim events fire and no
// sealed interim bubbles are created while streaming. After
// message.complete, the stored-history reconcile (sessions.changed →
// reconcileActiveTranscript) converges the view to the persisted
// transcript, which contains the mid-turn commentary as real assistant
// rows — exactly what a resume of this session would show. Pin that
// converged shape: ONE assistant message root for the whole turn…
await expect
.poll(
() => countAssistantMessageRoots(page),
{ timeout: 15_000, message: 'the settled turn should render as one assistant message' },
)
.toBe(1)
// …containing every commentary text and the final text exactly once.
for (const text of [...INTERIM_TEXTS.interims, INTERIM_TEXTS.finalText]) {
await expect
.poll(
() => countTranscriptMessagesContaining(page, text),
{ timeout: 15_000, message: `"${text}" should appear exactly once in the converged turn` },
)
.toBe(1)
}
})
})
@@ -46,6 +46,27 @@ test('renderer loads and shows DOM content', async () => {
expect(childCount).toBeGreaterThan(0)
})
test('boots to the app UI, not the QueryClient error boundary (#95560)', async () => {
const page = fixture!.page
await page.waitForSelector('#root', { state: 'attached', timeout: 30_000 })
// Wait until the root has real content (boot overlay fades, app paints) —
// the error boundary also paints, so assert on its absence explicitly.
await page.waitForFunction(
() => (document.getElementById('root')?.textContent ?? '').trim().length > 0,
undefined,
{ timeout: 60_000 },
)
const text = await page.locator('#root').textContent()
// The #95560 crash: a duplicate @tanstack/react-query runtime made the
// QueryClientProvider's context invisible to useQuery, so the app hit the
// error boundary at launch. Neither the boundary headline nor the throw
// message may appear on a healthy boot.
expect(text).not.toContain('No QueryClient set')
expect(text).not.toContain('Something broke in the interface')
})
test('HUD composer remains fully inside the transparent window', async () => {
const hudPagePromise = fixture!.app.waitForEvent('window')
@@ -57,6 +57,23 @@ test.describe('session compression', () => {
await send(page, 'E2E_COMPRESSION_THIRD')
await expect.poll(() => receivedUserTexts().filter(text => text === 'E2E_COMPRESSION_THIRD').length).toBe(1)
// The mock receiving the third prompt does not mean the TURN is over —
// /compress on a busy session errors with "session busy — /interrupt the
// current turn before /compress". Wait for the third reply to render and
// for the composer to leave its busy state (no Stop affordance) first.
await page.waitForFunction(
expected =>
((document.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').split(expected).length - 1) >= 3,
reply,
{ timeout: 90_000 }
)
await expect
.poll(
() => page.locator('[data-slot="composer-root"] button[aria-label="Stop"]').count(),
{ timeout: 30_000, message: 'turn should settle before /compress' }
)
.toBe(0)
// This test covers compression and continuation, not slash completion.
// Insert the complete command atomically and click Send so an async
// completion response cannot consume Enter as a picker acceptance.
@@ -89,6 +106,8 @@ test.describe('session compression in progress', () => {
protect_first_n: 0
protect_last_n: 1
auxiliary:
title_generation:
enabled: false
compression:
provider: custom
model: mock-model`,
@@ -124,7 +143,11 @@ auxiliary:
await expect(page.getByRole('status', { name: 'Summarizing thread' }).last()).toBeVisible()
const primary = page.locator('[data-slot="composer-root"] button[type="submit"]')
await expect(primary).toHaveAttribute('aria-label', 'Queue message')
// Since "running is not busy" (3bc52fb9df) an empty composer mid-turn
// shows Stop — the Queue affordance appears once a payload is typed, and
// the Enter path below still queues instead of steering while compaction
// holds the turn.
await expect(primary).toHaveAttribute('aria-label', 'Stop')
await send(page, queued)
await expect(page.getByText('1 Queued')).toBeVisible()
+76 -33
View File
@@ -30,6 +30,18 @@ const SESSION_RUNNING_DOT_LABEL = 'Session running'
/** Finished-unread dot aria-label. */
const UNREAD_DOT_LABEL = 'Finished — unread'
/**
* The auto-title auxiliary call hits the SAME mock provider as the chat turn,
* and its request carries the user's message — trigger keyword included. The
* mock's trigger matching is text-based, so the title call consumes a script
* index: the real chat turn then gets turn 2 (final answer, NO tool calls),
* the background process is never spawned, and the bg dot never appears.
* Whether that happens depends on which request lands first — the CI flake
* these specs had. Disable auto-title so script indices line up with real
* chat turns (same fix as interim-messages.spec.ts).
*/
const DISABLE_AUTO_TITLE = 'auxiliary:\n title_generation:\n enabled: false'
/** Send a message and wait for the final response to appear. */
async function sendMessageAndWait(
page: Page,
@@ -67,7 +79,7 @@ test.describe('sidebar states — background process and subagent', () => {
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend()
fixture = await setupMockBackend({ extraConfig: DISABLE_AUTO_TITLE })
await waitForAppReady(fixture, 120_000)
})
@@ -120,22 +132,32 @@ test.describe('sidebar states — subagent and background dot coexist', () => {
test.describe.configure({ mode: 'serial' })
let fixture: MockBackendFixture
// Hold the background process open until the test releases it. Without the
// sentinel the process is a bare `sleep 5` racing the agent turn (two model
// trips + a real subagent spawn): on a loaded runner the turn outlives the
// sleep, the process is reaped mid-turn, and the dot never appears at all —
// the CI flake this spec had.
const bgRelease = createBackgroundReleaseHandle()
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend()
fixture = await setupMockBackend({
extraConfig: DISABLE_AUTO_TITLE,
mockServer: { backgroundReleasePath: bgRelease.path },
})
await waitForAppReady(fixture, 120_000)
})
test.afterAll(async () => {
bgRelease.release()
await fixture?.cleanup()
bgRelease.cleanup()
})
test('background dot visible while subagent runs', async () => {
const page = fixture.page
// Start the turn but DON'T wait for the final answer yet — we want
// to assert the background dot is visible WHILE the subagent runs.
// Start the turn — a held background process plus a real subagent.
const composer = page.locator('[contenteditable="true"]').first()
await composer.waitFor({ state: 'visible', timeout: 10_000 })
await composer.click()
@@ -149,29 +171,45 @@ test.describe('sidebar states — subagent and background dot coexist', () => {
{ timeout: 15_000 },
)
// The background process (sleep 5) should show a "Background task
// running" dot while the subagent is also running.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should appear while subagent runs' },
)
.toBeGreaterThan(0)
// Evidence: the background dot is visible while the subagent runs.
await page.screenshot({ path: 'test-results/bg-dot-while-subagent-runs.png' })
// Now wait for the final answer to appear.
// While the turn is busy the dot-state priority paints the session as
// "working" ('Session running') — that claim OUTRANKS 'background', so
// polling for the bg dot mid-turn races the turn length against the poll
// budget. Wait for the turn to END (final text + running dot cleared),
// then assert the background dot as a stable, sentinel-held state.
await page.waitForFunction(
(text) => (document.body.textContent ?? '').includes(text),
SIDEBAR_CROSS_TEXTS.finalText,
{ timeout: 90_000 },
)
await expect
.poll(
() => page.locator(`[aria-label="${SESSION_RUNNING_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'session running dot should disappear after turn completes' },
)
.toBe(0)
// After the turn + auto-dismiss, the background dot should be gone.
await page.waitForTimeout(8000)
const bgCount = await page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count()
expect(bgCount, 'background dot should be gone after process exits').toBe(0)
// The background process is held open by the sentinel, so the bg dot is
// a stable state — poll only to absorb the event-driven flip landing a
// tick after the running dot clears.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should be visible after turn completes' },
)
.toBeGreaterThan(0)
// Evidence: the background dot is visible while the process runs.
await page.screenshot({ path: 'test-results/bg-dot-while-subagent-runs.png' })
// Release the process; the dot should clear on the completion event —
// event-driven, not a fixed sleep.
bgRelease.release()
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should be gone after process exits' },
)
.toBe(0)
})
})
@@ -190,6 +228,7 @@ test.describe('sidebar states — cross-session dot transition', () => {
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend({
extraConfig: DISABLE_AUTO_TITLE,
mockServer: { backgroundReleasePath: bgRelease.path },
})
await waitForAppReady(fixture, 120_000)
@@ -213,14 +252,13 @@ test.describe('sidebar states — cross-session dot transition', () => {
await composer.type('E2E_SIDEBAR_CROSS', { delay: 20 })
await page.keyboard.press('Enter')
// Wait for the background dot to appear.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should appear' },
)
.toBeGreaterThan(0)
// While the turn is busy the dot-state priority paints the session as
// "working" ('Session running') — that claim OUTRANKS 'background', so
// polling for the bg dot mid-turn races the turn length (two model trips
// + a real subagent spawn) against the poll budget: the CI flake this
// spec had. Wait for the turn to END first, then assert the bg dot as a
// stable, sentinel-held state.
//
// The final answer text streams before message.complete, so text visibility
// alone is not a completion barrier. Wait for the foreground-running state
// to clear before asserting the background-process state.
@@ -236,11 +274,16 @@ test.describe('sidebar states — cross-session dot transition', () => {
)
.toBe(0)
// The background dot must still be visible: the turn is done but the
// The background dot must be visible now: the turn is done but the
// process is held open by the sentinel, so this is a stable state rather
// than a window we have to catch in time.
const bgDuringTurn = await page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count()
expect(bgDuringTurn, 'background dot should still be visible after turn completes').toBeGreaterThan(0)
// than a window we have to catch in time. Poll to absorb the event-driven
// flip landing a tick after the running dot clears.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should be visible after turn completes' },
)
.toBeGreaterThan(0)
// Evidence: bg dot visible on session A while its turn is done but the
// background process hasn't exited yet.
+38 -14
View File
@@ -36,6 +36,18 @@ const BG_DOT_LABEL = 'Background task running'
/** Foreground turn-running dot aria-label. */
const SESSION_RUNNING_DOT_LABEL = 'Session running'
/**
* The auto-title auxiliary call hits the SAME mock provider as the chat turn,
* and its request carries the user's message — trigger keyword included. The
* mock's trigger matching is text-based, so the title call consumes a script
* index: the real chat turn then gets turn 2 (final answer, NO tool calls),
* the background process is never spawned, and the bg dot never appears.
* Whether that happens depends on which request lands first — the CI flake
* this spec had. Disable auto-title so script indices line up with real chat
* turns (same fix as interim-messages.spec.ts).
*/
const DISABLE_AUTO_TITLE = 'auxiliary:\n title_generation:\n enabled: false'
/** Locate a session's sidebar row by its preview text. */
function sessionRow(page: import('@playwright/test').Page, text: string) {
return page.locator('[data-slot="sidebar"] button').filter({ hasText: text }).first()
@@ -59,13 +71,13 @@ async function startTurnAndSwitchAway(page: import('@playwright/test').Page) {
{ timeout: 15_000 },
)
// Wait for the background dot — confirms the turn is running.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should appear' },
)
.toBeGreaterThan(0)
// NOTE: while the turn is busy the dot-state priority paints the session as
// "working" ('Session running'), which OUTRANKS the background claim — the
// 'Background task running' dot only appears once the turn completes while
// the (sentinel-held) process is still alive. Polling for the bg dot mid-turn
// races the turn length (two model trips + a real subagent spawn) against
// the poll budget, which is exactly the flake this spec had on CI. So: wait
// for the turn to END first, then assert the bg dot as a stable state.
// The final answer text streams before message.complete, so text visibility
// alone is not a completion barrier. Wait for the foreground-running state
@@ -82,11 +94,17 @@ async function startTurnAndSwitchAway(page: import('@playwright/test').Page) {
)
.toBe(0)
// The background dot must still be visible: the turn is done but the
// process is held open by the sentinel, so this is a stable state rather
// than a window we have to catch in time.
const bgDuringTurn = await page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count()
expect(bgDuringTurn, 'background dot should still be visible after turn completes').toBeGreaterThan(0)
// The background dot must be visible now: the turn is done but the process
// is held open by the sentinel, so this is a stable state rather than a
// window we have to catch in time. Poll rather than sampling once — the
// dot flip is event-driven off the busy=false publish and can land a tick
// after the running dot clears.
await expect
.poll(
() => page.locator(`[aria-label="${BG_DOT_LABEL}"]`).count(),
{ timeout: 30_000, message: 'background dot should be visible after turn completes' },
)
.toBeGreaterThan(0)
// Switch to a new session — session A is no longer $selectedStoredSessionId.
// This is required: openSessionTile bails if the session is already selected.
@@ -121,6 +139,7 @@ test.describe('sidebar states — tab (hidden) unread is correct', () => {
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend({
extraConfig: DISABLE_AUTO_TITLE,
mockServer: { backgroundReleasePath: bgRelease.path },
})
await waitForAppReady(fixture, 120_000)
@@ -142,7 +161,10 @@ test.describe('sidebar states — tab (hidden) unread is correct', () => {
// ⌃-click opens the session as a TAB (center dock = stacked, not visible
// unless it's the active tab). The session is NOT on screen.
const row = sessionRow(page, SIDEBAR_CROSS_TEXTS.finalText)
//
// With auto-title disabled the sidebar row is titled by the user's
// message (the trigger keyword), not the assistant's final text.
const row = sessionRow(page, 'E2E_SIDEBAR_CROSS')
await row.click({ modifiers: ['Control'] })
await page.waitForTimeout(2000)
@@ -182,6 +204,7 @@ test.describe.skip('sidebar states — split (visible) unread bug (RED)', () =>
test.beforeAll(async () => {
restartMockServer()
fixture = await setupMockBackend({
extraConfig: DISABLE_AUTO_TITLE,
mockServer: { backgroundReleasePath: bgRelease.path },
})
await waitForAppReady(fixture, 120_000)
@@ -204,7 +227,8 @@ test.describe.skip('sidebar states — split (visible) unread bug (RED)', () =>
// Drag the session row from the sidebar to the right edge of the workspace
// zone to create a SPLIT (side-by-side) tile. This triggers the real
// startSessionDrag → onCommit → openSessionTile(id, 'right', anchor) path.
const row = sessionRow(page, SIDEBAR_CROSS_TEXTS.finalText)
// With auto-title disabled the sidebar row is titled by the user's message.
const row = sessionRow(page, 'E2E_SIDEBAR_CROSS')
const rowBox = await row.boundingBox()
expect(rowBox, 'session row must be visible').not.toBeNull()
+21 -4
View File
@@ -10,7 +10,7 @@
* `syncSessionStateToView` to fire a second `setMessages` — a visual
* flicker as the transcript DOM was updated.
*
* This test pre-seeds a 32-message session into state.db, boots the app,
* This test pre-seeds a session into state.db, boots the app,
* clicks the session (cold resume — populates the warm cache), navigates
* away to a new chat, then clicks back (warm resume). Two detectors run:
*
@@ -50,8 +50,16 @@ const SESSION_TITLE = 'E2E Warm Resume Jitter Test'
// renderer's keep-alive visibility policy instead of relying on DOM order.
const SURFACE = '[data-composer-target]:not([data-pane-hidden] [data-composer-target])'
const ALL_SURFACES = '[data-composer-target]'
/** 32 messages (16 user/assistant pairs) — enough DOM churn for detection. */
const MESSAGE_COUNT = 32
/**
* 16 messages (8 user/assistant pairs) — enough DOM churn for detection while
* still fitting a hot-hidden pane's retention budget. A kept-alive pane keeps
* only its live tail (HIDDEN_TRANSCRIPT_RENDER_BUDGET = 40 weight units in
* thread/list.tsx); 16 short messages ≈ 32 units, so the whole transcript
* survives hiding. Above the budget, reveal legitimately backfills trimmed
* turns (additive DOM bursts) — that is paging, not the repaint bug this
* suite hunts, and it would drown the detectors.
*/
const MESSAGE_COUNT = 16
/** Seeded PRNG so the generated content is deterministic across runs. */
const RNG_SEED = 42
@@ -174,7 +182,16 @@ async function installRenderCounter(
: surfaces.at(-1)
const viewport = surface?.querySelector('[data-slot="aui_thread-viewport"]')
if (!viewport) {
throw new Error('Thread viewport not found before warm resume')
const diag = [...document.querySelectorAll(allSelector)].map(s => ({
hidden: Boolean(s.closest('[data-pane-hidden]')),
target: s.getAttribute('data-composer-target'),
hasViewport: Boolean(s.querySelector('[data-slot="aui_thread-viewport"]')),
textLen: (s.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').length,
head: (s.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').slice(0, 80),
tail: (s.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').slice(-80),
includesExpected: expected ? (s.querySelector('[data-slot="aui_thread-viewport"]')?.textContent ?? '').includes(expected) : null,
}))
throw new Error('Thread viewport not found before warm resume DIAG=' + JSON.stringify(diag) + ' expected=' + expected)
}
const state = { bursts: 0, mutations: 0, timeline: [] as number[], stopped: false, reconciles: 0 }
+5 -2
View File
@@ -66,9 +66,12 @@ test('processStartMarker resolves a real marker for the current process', async
assert.match(marker, /^(linux|win|winms|ps):.+/)
})
test('processStartMarker rejects for a PID that does not exist', async () => {
test('a missing PID is classified as ESRCH so reapOrphans can drop the record', async () => {
// Largest PIDs are bounded well below this on every supported platform.
await assert.rejects(processStartMarker(2 ** 30 + 12345))
// Windows Get-Process / macOS `ps -p` used to surface exit code 1, which
// the identity matchers treated as "unknown" and kept forever. The native
// gate throws ESRCH — the errno those catch blocks already map to gone.
await assert.rejects(processStartMarker(2 ** 30 + 12345), (error: NodeJS.ErrnoException) => error?.code === 'ESRCH')
})
// --- PID-only marker helpers --------------------------------------------------
+10
View File
@@ -21,6 +21,7 @@ import { execFile } from 'node:child_process'
import fs from 'node:fs'
import { electronProcessStartMarker } from './parent-process-identity'
import { isPidAlive } from './update-marker'
import { hiddenWindowsChildOptions } from './windows-child-options'
export function execText(command: string, args: string[], { timeout = 3000 } = {}): Promise<string> {
@@ -42,6 +43,15 @@ export function execText(command: string, args: string[], { timeout = 3000 } = {
* `claimDecision` / `probeStartMarker`).
*/
export async function processStartMarker(pid: number): Promise<string> {
// Cheap native dead-PID gate. Windows Get-Process / macOS `ps -p` exit 1
// on a missing PID (not ESRCH), so the identity matchers used to keep the
// orphan and re-probe it every launch (#92875). ESRCH is the code those
// catch blocks already map to "gone". Alive or uninspectable (EPERM) PIDs
// still fall through to the platform probe.
if (!isPidAlive(pid)) {
throw Object.assign(new Error(`PID ${pid} no longer exists`), { code: 'ESRCH' })
}
if (process.platform === 'linux') {
const stat = await fs.promises.readFile(`/proc/${pid}/stat`, 'utf8')
@@ -260,3 +260,62 @@ test('exit-before-announcement error stays clean when no output was buffered', a
return true
})
})
// ---------------------------------------------------------------------------
// bufferedOutput (#60323): a sentinel consumed BEFORE the wait attaches must
// still resolve. main.ts attaches an output tail at spawn, then awaits
// claimBackendChild/advanceBootProgress before calling this wait; flowing-mode
// stdout never replays consumed chunks to late listeners.
// ---------------------------------------------------------------------------
test('resolves from bufferedOutput when the sentinel was consumed before the wait attached (#60323)', async () => {
const child = makeFakeChild()
// Simulate the spawn-time output tail: it consumed the READY line already,
// and no further stdout data will ever arrive.
const alreadyConsumed = 'boot noise\nHERMES_BACKEND_READY port=43211\n'
const port = await waitForDashboardPortAnnouncement(child, {
bufferedOutput: () => alreadyConsumed,
timeoutMs: 500
})
assert.equal(port, 43211)
})
test('bufferedOutput accepts the legacy HERMES_DASHBOARD_READY sentinel too', async () => {
const child = makeFakeChild()
const port = await waitForDashboardPortAnnouncement(child, {
bufferedOutput: () => 'HERMES_DASHBOARD_READY port=43212\n',
timeoutMs: 500
})
assert.equal(port, 43212)
})
test('bufferedOutput without a sentinel still resolves from later live stdout', async () => {
const child = makeFakeChild()
const wait = waitForDashboardPortAnnouncement(child, {
bufferedOutput: () => 'uvicorn still importing...\n',
timeoutMs: 1000
})
child.stdout.emit('data', Buffer.from('HERMES_BACKEND_READY port=43213\n'))
assert.equal(await wait, 43213)
})
test('bufferedOutput without a sentinel still times out (no false positive)', async () => {
const child = makeFakeChild()
const wait = waitForDashboardPort(
child,
50,
() => '',
() => 'no sentinel here\n'
)
await assert.rejects(wait, /Timed out waiting/)
})
+36 -2
View File
@@ -51,8 +51,21 @@ function resolvePortAnnounceTimeoutMs(env = process.env) {
* on every terminal path — resolve, reject, or timeout — so repeated
* backend spawns don't leak listener slots on the child.
*/
function waitForDashboardPort(child, timeoutMs = resolvePortAnnounceTimeoutMs(), describeOutputTail = () => '') {
function waitForDashboardPort(
child,
timeoutMs = resolvePortAnnounceTimeoutMs(),
describeOutputTail = () => '',
bufferedOutput: () => string = () => ''
) {
return new Promise((resolve, reject) => {
// Seed the line buffer with any output the spawn-time tail already
// consumed (#60323): main.ts attaches its output tail at spawn, then
// awaits claimBackendChild + advanceBootProgress BEFORE this listener
// attaches. child.stdout is in flowing mode from the tail's listener, so
// a READY line flushed during that window is emitted once and never
// replayed to late listeners — the wait then times out at 90s and a
// healthy backend is killed. Scanning the tail's buffer (and seeding any
// trailing partial line) makes the listener-attach ordering irrelevant.
let buf = ''
let done = false
@@ -104,6 +117,19 @@ function waitForDashboardPort(child, timeoutMs = resolvePortAnnounceTimeoutMs(),
child.stdout.on('data', onData)
child.on('exit', onExit)
child.on('error', onError)
// Listener is live — now recover a sentinel that was already flushed and
// consumed before this promise existed. The snapshot is taken AFTER the
// listener attaches, so no chunk can fall between snapshot and listener.
if (!done) {
const alreadyBuffered = bufferedOutput()
const m = alreadyBuffered ? alreadyBuffered.match(_READY_RE) : null
if (m) {
cleanup()
resolve(parseInt(m[1], 10))
}
}
})
}
@@ -187,6 +213,14 @@ function waitForDashboardReadyFile(
function waitForDashboardPortAnnouncement(
child,
options: {
/**
* Returns the child's output buffered since SPAWN (the output tail's
* accumulated text, #60323). Scanned for an already-emitted READY
* sentinel so attaching this wait AFTER other awaits (backend claim,
* boot-progress IPC) can never lose the announcement: flowing-mode
* stdout never replays chunks to late listeners.
*/
bufferedOutput?: () => string
/** Returns a formatted stdout/stderr tail suffix for exit errors (#93608). */
describeOutputTail?: () => string
readyFile?: fs.PathOrFileDescriptor | null
@@ -200,7 +234,7 @@ function waitForDashboardPortAnnouncement(
return waitForDashboardReadyFile(options.readyFile, child, timeoutMs, describeOutputTail)
}
return waitForDashboardPort(child, timeoutMs, describeOutputTail)
return waitForDashboardPort(child, timeoutMs, describeOutputTail, options.bufferedOutput ?? (() => ''))
}
export {
@@ -0,0 +1,92 @@
import { describe, expect, it, vi } from 'vitest'
import { recycleOwnedBackend, recycleOwnedBackendTarget } from './backend-recycle'
describe('recycleOwnedBackendTarget', () => {
it('treats an empty or matching profile as the primary backend', () => {
expect(recycleOwnedBackendTarget(undefined, 'default')).toBe('primary')
expect(recycleOwnedBackendTarget('', 'default')).toBe('primary')
expect(recycleOwnedBackendTarget('default', 'default')).toBe('primary')
})
it('treats any other named profile as a pooled backend', () => {
expect(recycleOwnedBackendTarget('paid-ads', 'default')).toBe('pool')
})
})
describe('recycleOwnedBackend', () => {
it('kills the owned SSH serve before the primary child, then notifies apply', async () => {
const events: string[] = []
const target = await recycleOwnedBackend({
notifyApplied: () => events.push('applied'),
primaryProfile: 'default',
profile: undefined,
teardownPool: async () => {
events.push('pool')
},
teardownPrimary: async () => {
events.push('primary')
},
teardownSsh: async profile => {
events.push(`ssh:${profile}`)
}
})
expect(target).toBe('primary')
expect(events).toEqual(['ssh:', 'primary', 'applied'])
})
it('recycles a pooled profile without tearing down the primary', async () => {
const events: string[] = []
const target = await recycleOwnedBackend({
notifyApplied: () => events.push('applied'),
primaryProfile: 'default',
profile: 'paid-ads',
teardownPool: async profile => {
events.push(`pool:${profile}`)
},
teardownPrimary: async () => {
events.push('primary')
},
teardownSsh: async profile => {
events.push(`ssh:${profile}`)
}
})
expect(target).toBe('pool')
expect(events).toEqual(['ssh:paid-ads', 'pool:paid-ads'])
})
it('awaits SSH teardown before the local child even when SSH is slow', async () => {
const events: string[] = []
let releaseSsh!: () => void
const sshGate = new Promise<void>(resolve => {
releaseSsh = resolve
})
const run = recycleOwnedBackend({
notifyApplied: () => events.push('applied'),
primaryProfile: 'default',
teardownPool: vi.fn(),
teardownPrimary: async () => {
events.push('primary')
},
teardownSsh: async () => {
events.push('ssh-start')
await sshGate
events.push('ssh-done')
}
})
await Promise.resolve()
expect(events).toEqual(['ssh-start'])
releaseSsh()
await run
expect(events).toEqual(['ssh-start', 'ssh-done', 'primary', 'applied'])
})
})
+47
View File
@@ -0,0 +1,47 @@
/**
* Recycle a Desktop-owned backend after a code-skew 503.
*
* Closing the local tunnel/child is not enough for SSH: `serve --isolated`
* detaches with setsid/nohup, so a reconnect would reuse the still-alive
* stale process via the lockfile. Kill the owned remote serve first (while
* the SSH channel can still exec), then tear down the local child — the
* same order as connection apply (#97046, #91668).
*/
export type RecycleOwnedBackendTarget = 'pool' | 'primary'
export interface RecycleOwnedBackendDeps {
notifyApplied: () => void
primaryProfile: string
profile?: null | string
teardownPool: (profile: string) => Promise<void>
teardownPrimary: () => Promise<void>
teardownSsh: (profile: string) => Promise<void>
}
export function recycleOwnedBackendTarget(
profile: null | string | undefined,
primaryProfile: string
): RecycleOwnedBackendTarget {
const key = String(profile ?? '').trim()
return !key || key === primaryProfile ? 'primary' : 'pool'
}
export async function recycleOwnedBackend(deps: RecycleOwnedBackendDeps): Promise<RecycleOwnedBackendTarget> {
const target = recycleOwnedBackendTarget(deps.profile, deps.primaryProfile)
const profile = String(deps.profile ?? '').trim()
if (target === 'primary') {
await deps.teardownSsh('')
await deps.teardownPrimary()
deps.notifyApplied()
return target
}
await deps.teardownSsh(profile)
await deps.teardownPool(profile)
return target
}
+247 -14
View File
@@ -1,6 +1,11 @@
import { describe, expect, it } from 'vitest'
import { execFileSync } from 'node:child_process'
import { mkdirSync, mkdtempSync, rmSync, writeFileSync } from 'node:fs'
import { tmpdir } from 'node:os'
import { dirname, join } from 'node:path'
import { detectBundleSkew, isFallbackCommit, type RunGit } from './bundle-skew'
import { afterAll, describe, expect, it } from 'vitest'
import { detectBundleSkew, isFallbackCommit, type RunGit, RUNTIME_PATHS } from './bundle-skew'
const REPO = '/repo'
const STAMP = { commit: 'a'.repeat(40), source: 'ci' }
@@ -9,6 +14,36 @@ function gitReturning(stdout: string, code = 0): RunGit {
return async () => ({ code, stderr: '', stdout })
}
/**
* A git fake that answers per subcommand, so a test can say "ancestry fails,
* but the count would have claimed skew" — which is the shape of #92233.
*/
function gitAnswering(answers: Record<string, { code?: number; stderr?: string; stdout?: string }>): {
calls: string[][]
git: RunGit
} {
const calls: string[][] = []
const git: RunGit = async args => {
calls.push(args)
const answer = answers[args[0]] ?? {}
return {
code: answer.code ?? 0,
stderr: answer.stderr ?? '',
stdout: answer.stdout ?? ''
}
}
return { calls, git }
}
/** Every subcommand succeeds; rev-list reports `count`. */
function gitCounting(count: string): RunGit {
return gitAnswering({ 'merge-base': { code: 0 }, 'rev-list': { stdout: count } }).git
}
describe('isFallbackCommit', () => {
it('matches the all-zero placeholder at any stamp length', () => {
expect(isFallbackCommit('0'.repeat(40))).toBe(true)
@@ -19,27 +54,21 @@ describe('isFallbackCommit', () => {
describe('detectBundleSkew', () => {
it('reports stale when desktop commits landed after the stamp', async () => {
const result = await detectBundleSkew(STAMP, gitReturning('3\n'), REPO)
const result = await detectBundleSkew(STAMP, gitCounting('3\n'), REPO)
expect(result).toEqual({ desktopCommitsBehind: 3, outOfSync: true })
})
it('passes the stamp range scoped to apps/desktop', async () => {
let seen: string[] = []
const git: RunGit = async args => {
seen = args
return { code: 0, stderr: '', stdout: '0' }
}
it('counts only commits that touch runtime desktop paths', async () => {
const { calls, git } = gitAnswering({ 'merge-base': { code: 0 }, 'rev-list': { stdout: '0' } })
await detectBundleSkew(STAMP, git, REPO)
expect(seen).toEqual(['rev-list', '--count', `${STAMP.commit}..HEAD`, '--', 'apps/desktop'])
expect(calls[1]).toEqual(['rev-list', '--count', `${STAMP.commit}..HEAD`, '--', ...RUNTIME_PATHS])
})
it('is quiet when no desktop commits follow the stamp', async () => {
const result = await detectBundleSkew(STAMP, gitReturning('0\n'), REPO)
const result = await detectBundleSkew(STAMP, gitCounting('0\n'), REPO)
expect(result).toEqual({ desktopCommitsBehind: 0, outOfSync: false })
})
@@ -79,9 +108,213 @@ describe('detectBundleSkew', () => {
})
it('is quiet on unparsable rev-list output', async () => {
expect(await detectBundleSkew(STAMP, gitReturning('fatal: bad object'), REPO)).toEqual({
expect(await detectBundleSkew(STAMP, gitCounting('fatal: bad object'), REPO)).toEqual({
desktopCommitsBehind: null,
outOfSync: false
})
})
// #92233: a ZIP-fallback update rewrites the tree into a synthetic root, so
// the stamp commit still RESOLVES but is unreachable from HEAD. `A..HEAD`
// then counts HEAD's own history instead of measuring skew, and reports a
// permanent 1 even though apps/desktop is byte-identical. The user gets an
// "App build out of date" warning that cannot go off, so no remedy clears it.
it('is quiet when the stamp is not an ancestor of HEAD', async () => {
const { git } = gitAnswering({
'merge-base': { code: 1 },
'rev-list': { stdout: '1\n' }
})
expect(await detectBundleSkew(STAMP, git, REPO)).toEqual({
desktopCommitsBehind: null,
outOfSync: false
})
})
it('does not consult the commit count once ancestry is refused', async () => {
const { calls, git } = gitAnswering({
'merge-base': { code: 1 },
'rev-list': { stdout: '9999\n' }
})
await detectBundleSkew(STAMP, git, REPO)
expect(calls.map(args => args[0])).toEqual(['merge-base'])
})
it('asks about ancestry before counting, against the same stamp', async () => {
const { calls, git } = gitAnswering({
'merge-base': { code: 0 },
'rev-list': { stdout: '2\n' }
})
const result = await detectBundleSkew(STAMP, git, REPO)
expect(calls[0]).toEqual(['merge-base', '--is-ancestor', STAMP.commit, 'HEAD'])
expect(calls[1]?.[0]).toBe('rev-list')
expect(result).toEqual({ desktopCommitsBehind: 2, outOfSync: true })
})
it('is quiet when git cannot answer the ancestry question at all', async () => {
const { git } = gitAnswering({
'merge-base': { code: 128 },
'rev-list': { stdout: '4\n' }
})
expect(await detectBundleSkew(STAMP, git, REPO)).toEqual({
desktopCommitsBehind: null,
outOfSync: false
})
})
// Shallow clones, measured against git 2.55 rather than assumed. A stamp
// commit from BEFORE the graft boundary is not an object the clone has, so
// `--is-ancestor` exits 128 with "Not a valid object name" — the same
// unknowable bucket as any other missing commit, not a shallow-specific
// failure. A stamp INSIDE the shallow graph is answered normally, so
// `--fetch-depth`-limited CI checkouts do not lose skew detection wholesale;
// only builds stamped deeper than the checkout goes do.
it('is quiet on a shallow clone whose stamp predates the graft boundary', async () => {
const { calls, git } = gitAnswering({
'merge-base': {
code: 128,
stderr: `fatal: Not a valid object name ${STAMP.commit}`
},
'rev-list': { stdout: '7\n' }
})
expect(await detectBundleSkew(STAMP, git, REPO)).toEqual({
desktopCommitsBehind: null,
outOfSync: false
})
expect(calls).toHaveLength(1)
})
it('still detects skew on a shallow clone when the stamp is in the graph', async () => {
const { git } = gitAnswering({
'merge-base': { code: 0 },
'rev-list': { stdout: '2\n' }
})
expect(await detectBundleSkew(STAMP, git, REPO)).toEqual({
desktopCommitsBehind: 2,
outOfSync: true
})
})
})
// Real-git integration: proves the pathspec discriminates docs/e2e-only
// commits from runtime commits, and that a disconnected stamp goes quiet, in
// an actual repository rather than against a hand-written fake.
const scratchRepos: string[] = []
afterAll(() => {
for (const dir of scratchRepos) {
rmSync(dir, { force: true, recursive: true })
}
})
function scratchGit(repoRoot: string) {
return (...args: string[]) =>
execFileSync('git', ['-c', 'user.email=skew@test', '-c', 'user.name=skew', ...args], {
cwd: repoRoot,
stdio: ['ignore', 'pipe', 'pipe']
})
.toString()
.trim()
}
function makeScratchRepo(): { base: string; repoRoot: string } {
const repoRoot = mkdtempSync(join(tmpdir(), 'bundle-skew-'))
scratchRepos.push(repoRoot)
const git = scratchGit(repoRoot)
git('init', '-q', '-b', 'main')
git('commit', '-q', '--allow-empty', '-m', 'base')
return { base: git('rev-parse', 'HEAD'), repoRoot }
}
function writeFiles(repoRoot: string, files: string[]) {
for (const file of files) {
const target = join(repoRoot, file)
mkdirSync(dirname(target), { recursive: true })
writeFileSync(target, '')
}
}
function realGitRun(root: string): RunGit {
return async (args, options) => {
try {
const stdout = execFileSync('git', args, {
cwd: options.cwd || root,
stdio: ['ignore', 'pipe', 'pipe']
}).toString()
return { code: 0, stderr: '', stdout }
} catch (error) {
const e = error as { status?: number; stderr?: Buffer; stdout?: Buffer }
return {
code: e.status ?? 1,
stderr: e.stderr?.toString() ?? '',
stdout: e.stdout?.toString() ?? ''
}
}
}
}
describe('detectBundleSkew against a real git repo', () => {
it('is quiet when only docs and e2e specs changed under apps/desktop', async () => {
const { base, repoRoot } = makeScratchRepo()
const git = scratchGit(repoRoot)
writeFiles(repoRoot, ['apps/desktop/AGENTS.md', 'apps/desktop/e2e/boot.spec.ts'])
git('add', '.')
git('commit', '-q', '-m', 'docs and e2e only')
const result = await detectBundleSkew({ commit: base, source: 'local' }, realGitRun(repoRoot), repoRoot)
expect(result).toEqual({ desktopCommitsBehind: 0, outOfSync: false })
})
it('warns when a renderer file changed under apps/desktop', async () => {
const { base, repoRoot } = makeScratchRepo()
const git = scratchGit(repoRoot)
writeFiles(repoRoot, ['apps/desktop/src/app/new-feature.tsx', 'apps/desktop/README.md'])
git('add', '.')
git('commit', '-q', '-m', 'renderer change')
const result = await detectBundleSkew({ commit: base, source: 'local' }, realGitRun(repoRoot), repoRoot)
expect(result).toEqual({ desktopCommitsBehind: 1, outOfSync: true })
})
// The #92233 install, reproduced: the update rewrote the tree onto a fresh
// orphan root, so the stamp resolves but is unreachable. Real git answers
// `rev-list` with a positive count here — ancestry is the only thing that
// keeps the banner off.
it('is quiet when the stamp sits on a disconnected root', async () => {
const { base, repoRoot } = makeScratchRepo()
const git = scratchGit(repoRoot)
git('checkout', '-q', '--orphan', 'rewritten')
writeFiles(repoRoot, ['apps/desktop/src/app/shell.tsx'])
git('add', '.')
git('commit', '-q', '-m', 'synthetic root after a ZIP-fallback update')
const runGit = realGitRun(repoRoot)
// Precondition: the raw count this function used to trust is nonzero.
const raw = await runGit(['rev-list', '--count', `${base}..HEAD`, '--', ...RUNTIME_PATHS], { cwd: repoRoot })
expect(Number.parseInt(raw.stdout.trim(), 10)).toBeGreaterThan(0)
const result = await detectBundleSkew({ commit: base, source: 'local' }, runGit, repoRoot)
expect(result).toEqual({ desktopCommitsBehind: null, outOfSync: false })
})
})
+60 -11
View File
@@ -10,20 +10,33 @@
* Bot Mode update" reports).
*
* Detection: the packaged build carries install-stamp.json with the commit
* it was built from. If commits touching `apps/desktop/` exist in the source
* tree AFTER that stamp commit, the running renderer is provably missing
* desktop changes the installed runtime has:
* it was built from. If commits touching the RUNTIME paths of apps/desktop
* exist in the source tree AFTER that stamp commit, the running renderer is
* provably missing desktop changes the installed runtime has:
*
* git rev-list --count <stampCommit>..HEAD -- apps/desktop
* git merge-base --is-ancestor <stampCommit> HEAD
* git rev-list --count <stampCommit>..HEAD -- <RUNTIME_PATHS>
*
* Scoping to `apps/desktop/` keeps this quiet for the common case where the
* repo advances with agent-only changes — a shell built before those is not
* stale in any way the user can see.
* Ancestry has to come first, because `A..HEAD` only means "how far HEAD is
* ahead of A" when A is an ancestor of HEAD. When it is not, the range
* degenerates to HEAD's own history and the count stops describing skew at
* all: an update that rewrote the tree into a synthetic root leaves a stamp
* commit that still resolves but sits on a disconnected graph, so the count
* is a permanent >= 1 even when apps/desktop is byte-identical (#92233).
* Resolving the stamp is not enough — an unknown commit already exits
* non-zero below, but a merely *unrelated* one exits 0 with a positive count.
*
* Scoping to runtime paths keeps this quiet for the common cases where the
* repo advances without user-visible desktop changes: agent-only commits
* elsewhere in the repo, and docs / e2e spec / dev-script churn under
* apps/desktop that never reaches the shipped renderer or main process
* (#99832).
*
* Fail-quiet by design: no stamp (dev runs), a fallback all-zero stamp
* (non-git build), an unknown commit (stamp predates a shallow clone's
* history), or any git failure all report "not stale". This warning must
* never false-positive — it tells users their install is torn.
* history), a stamp that is not an ancestor of HEAD, or any git failure all
* report "not stale". This warning must never false-positive — it tells
* users their install is torn.
*
* Pure + injectable so it is testable without booting Electron or git.
*/
@@ -35,7 +48,7 @@ export interface BundleSkewStamp {
}
export interface BundleSkewResult {
/** Commits under apps/desktop/ between the build stamp and HEAD (null = unknowable). */
/** Runtime-path commits between the build stamp and HEAD (null = unknowable). */
desktopCommitsBehind: null | number
/** True only on positive proof that the renderer predates desktop changes in the tree. */
outOfSync: boolean
@@ -46,6 +59,23 @@ export type RunGit = (
options: { cwd: string }
) => Promise<{ code: number; stderr: string; stdout: string }>
/**
* The apps/desktop paths that actually reach the user: renderer sources,
* main-process sources, the HTML entry, the public/ assets Vite copies into
* the bundle, app icons, and the packaging config. Docs, e2e specs, scratch
* scripts, and dev tooling never reach the shipped app, so a delta confined
* to them is not a torn install in any way the user can see.
*/
export const RUNTIME_PATHS = [
'apps/desktop/src',
'apps/desktop/electron',
'apps/desktop/index.html',
'apps/desktop/public',
'apps/desktop/assets',
'apps/desktop/package.json',
'apps/desktop/vite.config.ts'
] as const
const NOT_STALE: BundleSkewResult = { desktopCommitsBehind: null, outOfSync: false }
/** Matches write-build-stamp.mjs's all-zero placeholder for non-git builds. */
@@ -63,7 +93,26 @@ export async function detectBundleSkew(
}
try {
const result = await runGit(['rev-list', '--count', `${stamp.commit}..HEAD`, '--', 'apps/desktop'], {
// Exit 0 = ancestor, 1 = unrelated or diverged, anything else = git could
// not answer (unknown object, shallow clone, not a repo). Only the first
// makes the commit count below a statement about skew, and the other two
// are the same "unknowable" the branches above already answer quietly.
//
// Deliberately not falling back to comparing apps/desktop CONTENT here.
// Differing content would prove the build and the tree disagree, but not
// which way round: a user sitting on an older checkout than their build
// would be told "app build out of date" backwards. Ancestry is what makes
// this a proof that the renderer PREDATES the tree, which is the claim the
// warning actually makes.
const ancestry = await runGit(['merge-base', '--is-ancestor', stamp.commit, 'HEAD'], {
cwd: repoRoot
})
if (ancestry.code !== 0) {
return NOT_STALE
}
const result = await runGit(['rev-list', '--count', `${stamp.commit}..HEAD`, '--', ...RUNTIME_PATHS], {
cwd: repoRoot
})
+49
View File
@@ -0,0 +1,49 @@
import { describe, expect, it } from 'vitest'
import { detectBundleSwap } from './bundle-swap'
const RUNNING = { builtAt: '2026-08-29T04:00:00.000Z', commit: 'a'.repeat(40), source: 'local' }
describe('detectBundleSwap', () => {
it('reports a swap when the on-disk stamp carries a different commit', () => {
const onDisk = { ...RUNNING, commit: 'b'.repeat(40) }
expect(detectBundleSwap(RUNNING, onDisk)).toBe(true)
})
it('reports a swap when the same commit was rebuilt (builtAt moved)', () => {
const onDisk = { ...RUNNING, builtAt: '2026-08-31T23:55:41.149Z' }
expect(detectBundleSwap(RUNNING, onDisk)).toBe(true)
})
// The Windows locked-binary case (#92233): the swap leg failed, so the
// bundle on disk is still the one we are running. A relaunch would repair
// nothing and cost the user their window.
it('is quiet when the on-disk stamp matches the running one', () => {
expect(detectBundleSwap(RUNNING, { ...RUNNING })).toBe(false)
})
it('is quiet without a running stamp (dev runs)', () => {
expect(detectBundleSwap(null, { ...RUNNING })).toBe(false)
})
it('is quiet without an on-disk stamp (unreadable resources)', () => {
expect(detectBundleSwap(RUNNING, null)).toBe(false)
})
it('is quiet on a fallback stamp on either side (non-git build)', () => {
const fallbackTagged = { ...RUNNING, source: 'fallback' }
const fallbackCommit = { ...RUNNING, commit: '0'.repeat(40) }
expect(detectBundleSwap(fallbackTagged, { ...RUNNING, commit: 'b'.repeat(40) })).toBe(false)
expect(detectBundleSwap(RUNNING, fallbackCommit)).toBe(false)
})
it('treats a missing builtAt on either side as unprovable at the same commit', () => {
const noBuiltAt = { commit: RUNNING.commit, source: 'local' }
expect(detectBundleSwap(noBuiltAt, { ...RUNNING })).toBe(false)
expect(detectBundleSwap(RUNNING, noBuiltAt)).toBe(false)
})
})
+61
View File
@@ -0,0 +1,61 @@
/**
* Swapped-bundle detection.
*
* The detached updater (scripts/desktop-update/posix.sh mac_swap /
* windows.ps1) rebuilds and swaps the packaged app on disk AFTER
* `hermes update` exits. An instance that was launched from the PRE-swap
* bundle — the user reopened Hermes mid-update, the #50238 gesture the boot
* gate exists for — would otherwise proceed to run the NEW runtime under the
* OLD renderer. The updater's own `open`/relaunch leg cannot rescue it: the
* single-instance lock turns that into a focus of the parked process, so no
* process ever loads the new build.
*
* That is the stale-renderer tail of a FULLY SUCCESSFUL update: the "App
* build out of date" banner appears right after the update, while the Updates
* card says "You're on the latest version" and so offers nothing that would
* clear it.
*
* Detection: compare the install stamp this process loaded at boot with the
* one on disk now. A different commit — or a different builtAt at the same
* commit (a dirty-tree or content-hash rebuild) — means the bundle under our
* feet is not the one we are running, and a plain relaunch loads it.
*
* Fail-quiet like bundle-skew: a missing stamp on either side (dev runs,
* unreadable resources) or a fallback all-zero commit reports "not swapped".
* This must never false-positive — a positive triggers an automatic relaunch.
*
* Pure so it is testable without booting Electron.
*/
import { isFallbackCommit } from './bundle-skew'
export interface BundleSwapStamp {
/** write-build-stamp.mjs build timestamp — differs on every rebuild. */
builtAt?: null | string
commit: string
/** write-build-stamp.mjs source tag — 'fallback' means the commit is fake. */
source?: null | string
}
/** True only on positive proof that the bundle on disk is not the running one. */
export function detectBundleSwap(running: BundleSwapStamp | null, onDisk: BundleSwapStamp | null): boolean {
if (!running?.commit || !onDisk?.commit) {
return false
}
if (running.source === 'fallback' || isFallbackCommit(running.commit)) {
return false
}
if (onDisk.source === 'fallback' || isFallbackCommit(onDisk.commit)) {
return false
}
if (running.commit !== onDisk.commit) {
return true
}
// Same commit: only a builtAt PRESENT ON BOTH sides can prove a rebuild —
// a missing timestamp (older stamp schema) proves nothing.
return Boolean(running.builtAt && onDisk.builtAt && running.builtAt !== onDisk.builtAt)
}
@@ -4,6 +4,7 @@ import {
applyConnectionChange,
commitConnectionFailure,
resolveTerminalConnection,
sshQuitShouldBlock,
teardownSshState
} from './connection-apply'
@@ -91,6 +92,48 @@ describe('resolveTerminalConnection', () => {
})
})
describe('sshQuitShouldBlock', () => {
it('waits when connections exist and teardown has not finished', () => {
expect(sshQuitShouldBlock({ teardownDone: false, connectionCount: 1, bootstrapPending: 0, inFlight: null })).toBe(
true
)
})
it('waits when bootstrap is still running', () => {
expect(sshQuitShouldBlock({ teardownDone: false, connectionCount: 0, bootstrapPending: 1, inFlight: null })).toBe(
true
)
})
it('waits when the map is empty but a remote kill is already in flight', () => {
expect(
sshQuitShouldBlock({
teardownDone: false,
connectionCount: 0,
bootstrapPending: 0,
inFlight: Promise.resolve()
})
).toBe(true)
})
it('does not block a second quit after teardown finished', () => {
expect(
sshQuitShouldBlock({
teardownDone: true,
connectionCount: 1,
bootstrapPending: 1,
inFlight: Promise.resolve()
})
).toBe(false)
})
it('does not block quit when there is nothing to tear down', () => {
expect(sshQuitShouldBlock({ teardownDone: false, connectionCount: 0, bootstrapPending: 0, inFlight: null })).toBe(
false
)
})
})
describe('teardownSshState', () => {
it('terminates the owned remote backend before closing its tunnel and SSH transport', async () => {
const events: string[] = []
+16
View File
@@ -61,6 +61,21 @@ async function resolveTerminalConnectionForSender(webContentsId, getTarget, ensu
)
}
/** A second before-quit must still wait for an in-flight remote kill.
*
* teardownSshConnection deletes the sshConnections entry first, then
* SSH-execs kill. backendShutdown's finally() calls app.quit() and
* re-enters before-quit with an empty map. Without `inFlight`, Electron
* exits while disconnect is running and the detached serve --isolated
* stays at pid 1 (post-#95085 leftover on #91668: window X on Windows). */
function sshQuitShouldBlock({ teardownDone, connectionCount, bootstrapPending, inFlight }) {
if (teardownDone) {
return false
}
return connectionCount > 0 || bootstrapPending > 0 || Boolean(inFlight)
}
async function teardownSshState(state, { cleanupRemote }) {
// Remote process first, while the SSH channel can still exec kill.
// Then drop the local forward and close the transport. Each step is
@@ -91,5 +106,6 @@ export {
commitConnectionFailure,
resolveTerminalConnection,
resolveTerminalConnectionForSender,
sshQuitShouldBlock,
teardownSshState
}
@@ -334,10 +334,10 @@ const ROUTES = [
expected: { backend: 'primary', descriptorProfile: null, scopePath: false }
},
{
name: 'a renamed primary profile still owns the window backend',
name: 'a renamed primary profile on a global remote is still scoped on the wire',
profile: ' coder ',
opts: { primaryProfile: 'coder', globalRemote: true },
expected: { backend: 'primary', descriptorProfile: null, scopePath: false }
expected: { backend: 'primary', descriptorProfile: 'coder', scopePath: true }
},
{
name: 'an unset profile resolves to the primary',
@@ -526,14 +526,14 @@ test('pathWithGlobalRemoteProfile appends profile in global remote mode', () =>
)
})
test('pathWithGlobalRemoteProfile skips the primary profile, which the remote already serves', () => {
test('pathWithGlobalRemoteProfile scopes the primary label because the dashboard launch home may differ', () => {
assert.equal(
pathWithGlobalRemoteProfile('/api/model/info', 'coder', {
globalRemote: true,
primaryProfile: 'coder',
profileRemoteOverride: false
}),
'/api/model/info'
'/api/model/info?profile=coder'
)
})
+15 -2
View File
@@ -592,6 +592,7 @@ const LOCAL_PRIMARY_SCOPED_ROUTES = new Set([
'GET /api/skills/content',
'PUT /api/skills/toggle',
'POST /api/skills/hub/install',
'GET /api/skills/hub/official',
'GET /api/skills/hub/preview',
'GET /api/skills/hub/scan',
'GET /api/skills/hub/search',
@@ -653,7 +654,9 @@ function localPrimaryRequestScope(opts: ProfileRouteOptions): boolean | null {
* The one place that answers "which backend serves profile P, and does its
* REST path need a profile scope?". Six routes, in precedence order:
*
* 1. The primary profile owns the window backend outright.
* 1. The primary profile owns a local/window backend outright; on a global
* remote its label is still carried per request because launch home can
* differ from the selected profile.
* 2. A profile with its own remote override gets a pooled descriptor for that
* host, which is already scoped to it.
* 3. A profile inheriting the app-global remote shares the primary backend —
@@ -673,10 +676,20 @@ function resolveProfileBackendRoute(profile, opts: ProfileRouteOptions = {}): Pr
const scopedProfile = connectionScopeKey(profile)
const primaryProfile = connectionScopeKey(opts.primaryProfile) || 'default'
if (!scopedProfile || scopedProfile === primaryProfile) {
if (!scopedProfile) {
return { backend: 'primary', descriptorProfile: null, scopePath: false }
}
if (scopedProfile === primaryProfile) {
// A global remote is a multi-profile dashboard, not a backend process
// launched for this Desktop label. Even its "primary" label must travel on
// the wire: the dashboard's process HERMES_HOME can belong to a different
// launch profile, so a bare request silently reads that profile instead.
return opts.globalRemote
? { backend: 'primary', descriptorProfile: scopedProfile, scopePath: true }
: { backend: 'primary', descriptorProfile: null, scopePath: false }
}
if (opts.profileRemoteOverride) {
return { backend: 'pool', descriptorProfile: null, scopePath: false }
}
@@ -1595,12 +1595,13 @@ test('drift heal respects a deliberate primary pick on a registered route', () =
assert.equal(drifted.registry.primary, LOCAL_CONNECTION_ID)
})
test('drift heal ignores local, ssh, and unparseable v1 routes', () => {
test('drift heal ignores local and unparseable v1 routes', () => {
const registry = emptyRegistry()
for (const v1 of [
{ mode: 'local', remote: {} },
{ mode: 'ssh', remote: { host: 'box' } },
{ mode: 'ssh', remote: {} },
{ mode: 'ssh', remote: { host: ' ' } },
{ mode: 'remote', remote: { url: 'not a url' } },
{ mode: 'remote', remote: {} },
null
@@ -1612,6 +1613,92 @@ test('drift heal ignores local, ssh, and unparseable v1 routes', () => {
}
})
test('drift heal registers a v1 SSH route the registry never learned about and makes it primary', () => {
// mgallmur-glitch's shape: registry migrated while local-only, then Settings
// pointed v1 at an SSH host (host, no url). The registry cannot name it, so
// primary stays 'local' and the files re-drift after every update relaunch.
const drifted = reconcileRegistryDrift(emptyRegistry(), {
mode: 'ssh',
remote: { host: 'devbox.example.com', user: 'omar', port: 2222 }
})
assert.equal(drifted.changed, true)
const ssh = drifted.registry.connections.find(connection => connection.kind === 'ssh')
assert.ok(ssh)
assert.equal(ssh.host, 'devbox.example.com')
assert.equal(ssh.user, 'omar')
assert.equal(ssh.port, 2222)
assert.equal(drifted.registry.primary, ssh.id)
assert.equal(drifted.registry.lastUsed, ssh.id)
// The whole point: the live v1 SSH descriptor can now be named.
assert.equal(
resolvedConnectionId(drifted.registry, {
mode: 'remote',
remoteKind: 'ssh',
ssh: { host: 'devbox.example.com', user: 'omar', port: 2222 }
}),
ssh.id
)
})
test('drift heal leaves a registry that already knows the v1 SSH route untouched', () => {
const first = reconcileRegistryDrift(emptyRegistry(), {
mode: 'ssh',
remote: { host: 'devbox.example.com', user: 'omar' }
})
assert.equal(first.changed, true)
const drifted = reconcileRegistryDrift(first.registry, {
mode: 'ssh',
remote: { host: 'DEVBOX.example.com', user: 'Omar' }
})
assert.equal(drifted.changed, false)
assert.equal(drifted.registry, first.registry)
})
test('drift heal respects a deliberate primary pick on a registered SSH route', () => {
let registry = reconcileRegistryDrift(emptyRegistry(), {
mode: 'ssh',
remote: { host: 'devbox.example.com' }
}).registry
registry = setPrimaryConnection(registry, LOCAL_CONNECTION_ID)
const drifted = reconcileRegistryDrift(registry, {
mode: 'ssh',
remote: { host: 'devbox.example.com' }
})
assert.equal(drifted.changed, false)
assert.equal(drifted.registry.primary, LOCAL_CONNECTION_ID)
})
test('drift heal adds the missing SSH source without disturbing other registered sources', () => {
let registry = emptyRegistry()
registry = upsertConnection(registry, {
id: 'homelab',
kind: 'remote',
label: 'Homelab',
url: 'https://homelab.example.com',
authMode: 'token',
token: { keep: true }
})
const drifted = reconcileRegistryDrift(registry, {
mode: 'ssh',
remote: { host: 'devbox.example.com' }
})
assert.equal(drifted.changed, true)
assert.ok(drifted.registry.connections.some(connection => connection.id === 'homelab'))
assert.ok(drifted.registry.connections.some(connection => connection.kind === 'ssh'))
})
test('drift heal adds the missing remote without disturbing other registered sources', () => {
let registry = emptyRegistry()
@@ -1496,6 +1496,13 @@ export function reconcileAppliedGlobalConnection(
* registry entry at all. That is the drift state and nothing else. If the
* route is already registered but `primary` names another source, the user
* chose that in the Connections panel and we leave it alone.
*
* SSH drifts the same way remote does: a v1 global `mode:'ssh'` route (host,
* no url) written by Settings after the one-shot migration has no registry
* identity, so `resolvedConnectionId` returns null, `primary` stays `local`,
* and every launch re-homes the window onto a local backend — and because the
* heal used to skip SSH entirely, the two files re-drifted after every update
* relaunch instead of converging once.
*/
export function reconcileRegistryDrift(
registry: ConnectionRegistry,
@@ -1504,6 +1511,60 @@ export function reconcileRegistryDrift(
const config = v1 && typeof v1 === 'object' ? (v1 as Record<string, any>) : {}
const unchanged = { changed: false, registry }
if (config.mode === 'ssh') {
const ssh = normalizeSshConfig({
...(config.remote && typeof config.remote === 'object' ? config.remote : {}),
mode: 'ssh'
})
if (!ssh) {
// A v1 SSH route without a usable host is not a route we can register.
return unchanged
}
const target = normalizedSshTarget(ssh)
const alreadyRegistered = registry.connections.some(
connection =>
connection.kind === 'ssh' &&
normalizedSshTarget(connection) === target &&
(connection.port ?? 22) === (ssh.port ?? 22)
)
if (alreadyRegistered) {
// Route is known; if primary names another source, that is the user's
// Connections-panel choice, not drift.
return unchanged
}
const { mode: _mode, ...sshFields } = ssh
let entry: RegistryConnection
try {
entry = normalizeConnectionInput(
{
kind: 'ssh',
label: uniqueLabel(
ssh.host,
registry.connections.map(connection => connection.label)
),
...sshFields
},
registry
)
} catch {
// Validation failure (e.g. a crafted collision) must not corrupt the
// registry; the v1 path keeps failing the way it already does.
return unchanged
}
return {
changed: true,
registry: { ...upsertConnection(registry, entry), primary: entry.id, lastUsed: entry.id }
}
}
if (!modeIsRemoteLike(config.mode)) {
return unchanged
}
@@ -0,0 +1,130 @@
import assert from 'node:assert/strict'
import { test } from 'vitest'
import {
GATEWAY_STOP_TIMEOUT_MS,
startGatewaysAfterUpdateAbort,
stopGatewayBeforeUpdate
} from './gateway-stop-before-update'
const CLI = 'C:\\Users\\x\\hermes\\hermes-agent\\venv\\Scripts\\hermes.exe'
const HOME = 'C:\\Users\\x\\hermes'
function fakeExec(ok: boolean) {
return (_command: string, _args: string[], _options: unknown) => {
if (!ok) {
throw new Error('spawn ENOENT')
}
return Buffer.from('')
}
}
test('non-Windows is a no-op and never invokes the CLI', () => {
const calls: Array<[string, string[]]> = []
const ran = stopGatewayBeforeUpdate(CLI, HOME, {
isWindows: false,
existsSync: () => true,
execFileSync: fakeExec(true) as never,
spy: (c, a) => calls.push([c, a])
})
assert.equal(ran, false)
assert.deepEqual(calls, [])
})
test('Windows with missing CLI shim returns false and does not exec', () => {
const calls: Array<[string, string[]]> = []
const ran = stopGatewayBeforeUpdate(CLI, HOME, {
isWindows: true,
existsSync: () => false,
execFileSync: fakeExec(true) as never,
spy: (c, a) => calls.push([c, a])
})
assert.equal(ran, false)
assert.deepEqual(calls, [[CLI, ['gateway', 'stop', '--all']]])
})
test('Windows with live CLI invokes "gateway stop --all" and returns true', () => {
let seenCommand = ''
let seenArgs: string[] = []
const ran = stopGatewayBeforeUpdate(CLI, HOME, {
isWindows: true,
existsSync: () => true,
execFileSync: ((command: string, args: string[]) => {
seenCommand = command
seenArgs = args
return Buffer.from('')
}) as never
})
assert.equal(ran, true)
assert.equal(seenCommand, CLI)
assert.deepEqual(seenArgs, ['gateway', 'stop', '--all'])
})
test('Windows with failing CLI returns false (best-effort, never throws)', () => {
const ran = stopGatewayBeforeUpdate(CLI, HOME, {
isWindows: true,
existsSync: () => true,
execFileSync: fakeExec(false) as never
})
assert.equal(ran, false)
})
test('passes a generous timeout with hidden console (taskkill window suppression)', () => {
let seenOptions: unknown
stopGatewayBeforeUpdate(CLI, HOME, {
isWindows: true,
existsSync: () => true,
execFileSync: ((_c: string, _a: string[], options: unknown) => {
seenOptions = options
return Buffer.from('')
}) as never
})
assert.deepEqual(seenOptions, {
timeout: GATEWAY_STOP_TIMEOUT_MS,
windowsHide: true,
stdio: 'ignore',
encoding: 'utf8'
})
})
test('abort-path counterpart invokes "gateway start --all" (drain-semantics restore)', () => {
let seenArgs: string[] = []
const ran = startGatewaysAfterUpdateAbort(CLI, {
isWindows: true,
existsSync: () => true,
execFileSync: ((_c: string, args: string[]) => {
seenArgs = args
return Buffer.from('')
}) as never
})
assert.equal(ran, true)
assert.deepEqual(seenArgs, ['gateway', 'start', '--all'])
})
test('abort-path counterpart is a no-op off Windows', () => {
const calls: Array<[string, string[]]> = []
const ran = startGatewaysAfterUpdateAbort(CLI, {
isWindows: false,
existsSync: () => true,
execFileSync: fakeExec(true) as never,
spy: (c, a) => calls.push([c, a])
})
assert.equal(ran, false)
assert.deepEqual(calls, [])
})
@@ -0,0 +1,97 @@
/**
* gateway-stop-before-update.ts
*
* Windows-only helper for the update hand-off (#70337): stop every
* separately-running messaging gateway BEFORE the venv-shim lock poll.
*
* Why not just tree-kill gateway.pid's PID:
* - gateway.pid records the uv WORKER process, but the venv shim lock is
* held by its parent LAUNCHER (venv\Scripts\python.exe). taskkill /T from
* the worker PID does not reach parents, so the lock could survive.
* - a single gateway.pid read misses multi-profile setups entirely.
*
* So we delegate to `hermes gateway stop --all`: the CLI discovers every
* profile's gateway processes (launcher + worker) via find_gateway_pids,
* drains in-flight agents (planned-stop marker -> resume_pending), and
* force-kills survivors — the same logic `hermes update`'s
* _pause_windows_gateways_for_update relies on.
*
* Pure + dependency-injected so the launcher/worker and multi-profile
* behavior is assertable without booting Electron.
*/
import { execFileSync, type ExecFileSyncOptionsWithStringEncoding } from 'node:child_process'
import fs from 'node:fs'
export interface StopGatewayBeforeUpdateDeps {
/** Defaults to process.platform === 'win32'; injectable for tests. */
isWindows?: boolean
/** Defaults to fs.existsSync; injectable for tests. */
existsSync?: (p: string) => boolean
/** Defaults to execFileSync from node:child_process; injectable for tests. */
execFileSync?: (command: string, args: string[], options: ExecFileSyncOptionsWithStringEncoding) => Buffer | string
/** Observability hook for tests. */
spy?: (command: string, args: string[]) => void
}
export const GATEWAY_STOP_TIMEOUT_MS = 20_000
/**
* Best-effort stop of all-profile messaging gateways via the CLI.
* Never throws: a wedged/absent CLI must not abort the update hand-off
* (the shim-lock poll + the updater's venv-blocker scan still fail loudly
* if the venv stays held). Returns true when the CLI ran (or was invoked
* with the injected spy), false when skipped (non-Windows / missing CLI).
*/
export function stopGatewayBeforeUpdate(
hermesCliPath: string,
hermesHome: string,
deps: StopGatewayBeforeUpdateDeps = {}
): boolean {
return runGatewayLifecycleCommand(hermesCliPath, ['gateway', 'stop', '--all'], deps)
}
/**
* Drain-semantics counterpart (#76057 review): `gateway stop --all` before
* the lock gate takes gateways down even when the update later ABORTS
* (venv-blocked by a user terminal, probe failure, updater spawn failure).
* The updater's own pause machinery resumes what it pauses — the Desktop
* must mirror that on its abort paths, or a failed update strands every
* profile's gateway stopped. Best-effort, never throws.
*/
export function startGatewaysAfterUpdateAbort(hermesCliPath: string, deps: StopGatewayBeforeUpdateDeps = {}): boolean {
return runGatewayLifecycleCommand(hermesCliPath, ['gateway', 'start', '--all'], deps)
}
function runGatewayLifecycleCommand(hermesCliPath: string, args: string[], deps: StopGatewayBeforeUpdateDeps): boolean {
const isWindows = deps.isWindows ?? process.platform === 'win32'
if (!isWindows) {
return false
}
const existsSync = deps.existsSync ?? fs.existsSync
const exec = deps.execFileSync ?? execFileSync
if (deps.spy) {
deps.spy(hermesCliPath, args)
}
if (!existsSync(hermesCliPath)) {
return false
}
try {
exec(hermesCliPath, args, {
timeout: GATEWAY_STOP_TIMEOUT_MS,
windowsHide: true,
stdio: 'ignore',
encoding: 'utf8'
})
return true
} catch {
// Best-effort (see header comment).
return false
}
}
+355 -29
View File
@@ -63,6 +63,7 @@ import {
verifyHermesCli
} from './backend-probes'
import { waitForDashboardPortAnnouncement } from './backend-ready'
import { recycleOwnedBackend } from './backend-recycle'
import { isPidAliveWindows, waitForBackendRelease } from './backend-release-gate'
import {
isHostKeyChangedBootFailure,
@@ -87,7 +88,8 @@ import {
buildBrowserWindowUrl
} from './browser-windows'
import { detectBundleSkew } from './bundle-skew'
import { applyConnectionChange, teardownSshState } from './connection-apply'
import { detectBundleSwap } from './bundle-swap'
import { applyConnectionChange, sshQuitShouldBlock, teardownSshState } from './connection-apply'
import {
apiRequestRegistryConnectionId,
authModeFromStatus,
@@ -196,6 +198,7 @@ import {
resolveGatewayFileBackend,
writeBufferToFile
} from './gateway-file-download'
import { startGatewaysAfterUpdateAbort, stopGatewayBeforeUpdate } from './gateway-stop-before-update'
import { probeGatewayWebSocket } from './gateway-ws-probe'
import { registerGitIpc } from './git-ipc'
import { clearStaleGitLocks } from './gitlock'
@@ -280,6 +283,7 @@ import { selectPoolEvictions } from './pool-eviction'
import { createPoolStopper } from './pool-stop'
import { poolTouchKeys } from './pool-touch-scope'
import { createKeepAwake } from './power-save'
import { capturePreviewContents } from './preview-capture'
import { PreviewReachRegistry } from './preview-reach'
import {
createPrimaryRemoteConnection,
@@ -296,6 +300,7 @@ import {
profileNameFromDeleteRequest,
resolveRouteProfile
} from './profile-delete-routing'
import { migrateActiveProfileIfMissing as migrateActiveProfileIfMissingPure } from './profile-migration'
import { prepareProfileRenameLifecycle, profileRenameFromRequest } from './profile-rename-routing'
import {
buildSidebarSessionSliceParams,
@@ -391,6 +396,7 @@ import {
scanVenvBlockers,
stopSafeVenvBlockers
} from './venv-blocker-scan'
import { isHermesOwnedVenvDaemon } from './venv-holder-select'
import { fetchMarketplaceThemes, searchMarketplaceThemes } from './vscode-marketplace'
import { createWakeIndicatorWindowController } from './wake-indicator-window'
import { enumerateWindowsFrontToBack, enumerationFailed, readWindowBelow } from './window-below'
@@ -2151,6 +2157,59 @@ function updateGateDeps() {
}
}
// One-shot guard for the automatic bundle-swap relaunch below: the relaunched
// instance carries this flag so a stamp that still mismatches (unreadable
// resources, exotic packaging) can never produce a relaunch loop.
const BUNDLE_SWAP_RELAUNCH_FLAG = '--hermes-bundle-swap-relaunched'
// How long the parked instance waits for its own scheduled exit to land before
// giving up and booting the stale build anyway. Better a torn renderer with a
// banner than a window that never comes back.
const BUNDLE_SWAP_RELAUNCH_FAILSAFE_MS = 15_000
// The detached updater swaps the packaged bundle on disk AFTER `hermes update`
// exits (posix.sh mac_swap / windows.ps1). An instance reopened mid-update —
// the #50238 gesture the gate above exists for — was launched from the
// PRE-swap bundle, and the updater's `open` leg then merely focuses us (single
// instance), so no process ever loads the new build. Letting boot proceed here
// runs the new runtime under the old renderer: exactly the skew
// detectRendererSkew() warns about, except the Updates card already says
// "latest", so the warning's own remedy has nothing to run.
//
// This is the earliest point where the swap is PROVABLE — it happens while we
// are parked on the gate, so checking any sooner (at `ready`, before the gate)
// only ever compares a stamp with itself. Relaunching here also keeps the
// boot-progress window up for the whole wait instead of leaving the user with
// no window at all.
//
// Returns true when the relaunch was scheduled; the caller must park rather
// than continue booting, because the process exits underneath it.
function relaunchIntoSwappedBundle() {
if (!IS_PACKAGED || process.argv.includes(BUNDLE_SWAP_RELAUNCH_FLAG)) {
return false
}
if (!detectBundleSwap(INSTALL_STAMP, loadInstallStamp())) {
return false
}
rememberLog('[updates] app bundle was swapped during the update; relaunching into the new build')
try {
app.relaunch({
args: [...buildNoSandboxRelaunchArgs(process.argv.slice(1)), BUNDLE_SWAP_RELAUNCH_FLAG]
})
} catch (err) {
rememberLog(`[updates] bundle-swap relaunch failed: ${err?.message || err}; continuing with the current build`)
return false
}
void exitAfterBackendShutdown(0)
return true
}
// Block until no live update is in progress (or we hit the wait timeout).
// Emits a boot-progress phase so the renderer shows "Update in progress…"
// rather than a frozen splash. Returns true if it parked at all.
@@ -2213,6 +2272,14 @@ async function waitForUpdateToFinish() {
if (outcome === 'timeout') {
rememberLog('[updates] update still in progress after wait timeout; starting backend anyway')
} else if (relaunchIntoSwappedBundle()) {
await advanceBootProgress('backend.update-restart', 'Restarting Hermes to load the updated app…', 14)
// Park while the scheduled exit lands so this stale build never starts a
// backend; the failsafe below only runs if the exit somehow does not.
await new Promise(resolve => setTimeout(resolve, BUNDLE_SWAP_RELAUNCH_FAILSAFE_MS))
rememberLog(
`[updates] relaunch did not land within ${BUNDLE_SWAP_RELAUNCH_FAILSAFE_MS}ms; continuing with the current build`
)
} else {
rememberLog('[updates] update finished; proceeding with backend start')
}
@@ -3222,6 +3289,54 @@ function isShimLocked(shimPath) {
}
}
// Kill only Hermes-OWNED venv daemons (the memory plugin's hindsight daemon:
// exe under venv\Scripts AND cmdline referencing hindsight_api.main). The
// daemon is spawned DETACHED, so it outlives the backend tree-kill and keeps
// venv files mapped. External holders (a user terminal running `hermes`,
// unrelated scripts) are NOT killed — scanVenvBlockers reports them and the
// hand-off aborts, per existing design. Selection lives in the pure
// venv-holder-select module (ordinal path-prefix, no PowerShell -like
// wildcard hazards) so it's testable without Electron.
function killHermesOwnedVenvDaemons(updateRoot) {
if (!IS_WINDOWS) {
return
}
const scriptsDir = path.join(updateRoot, 'venv', 'Scripts')
let holders = []
try {
const out = execFileSync(
'powershell',
[
'-NoProfile',
'-Command',
'Get-CimInstance Win32_Process | Where-Object { $_.ExecutablePath -and $_.CommandLine } | Select-Object ProcessId, ExecutablePath, CommandLine | ConvertTo-Json -Compress'
],
hiddenWindowsChildOptions({ encoding: 'utf8', stdio: ['ignore', 'pipe', 'ignore'], timeout: 15_000 })
)
const parsed = JSON.parse(String(out || '[]'))
holders = (Array.isArray(parsed) ? parsed : [parsed]).filter(p =>
isHermesOwnedVenvDaemon(p?.ExecutablePath, p?.CommandLine, scriptsDir)
)
} catch {
// Best-effort: the venv-blocker scan downstream is the real backstop.
return
}
for (const holder of holders) {
const pid = Number(holder?.ProcessId)
if (Number.isInteger(pid) && pid > 0) {
rememberLog(`[updates] stopping Hermes-owned venv daemon (hindsight) PID ${pid} before hand-off`)
forceKillProcessTree(pid)
}
}
}
// Force-kill the entire process TREE rooted at each PID. Node's child.kill()
// only signals the direct child, so on Windows a backend `hermes.exe` that
// spawned its own grandchildren (a `hermes` REPL, a pty terminal session, the
@@ -3572,6 +3687,27 @@ async function releaseBackendLock(updateRoot, tag) {
stopAllPoolBackends
})
// Stop separately-running messaging gateways (all profiles) BEFORE the
// release gate. The gateway is launched by the gateway-launcher desktop
// plugin via /api/gateway/start and is NOT in backendConnectionState or
// backendPool, so the tree-kills above never see it — on Windows its
// launcher (venv\Scripts\python.exe) keeps the venv mandatory-locked and
// the 15s gate aborts the hand-off before the venv-blocker scan's
// pausable-gateway exemption ever gets a chance (#70337). Delegate to
// `hermes gateway stop --all`: the CLI discovers every profile's gateway
// (launcher + worker — gateway.pid records only the uv WORKER, and
// taskkill /T from the worker never reaches its parent), drains in-flight
// agents, and force-kills survivors. Best-effort; abort paths restore via
// startGatewaysAfterUpdateAbort. No-op off Windows.
stopGatewayBeforeUpdate(venvHermesShimPath(updateRoot), HERMES_HOME)
// Reap Hermes-OWNED venv daemons the tree-kill above cannot reach: the
// memory plugin's hindsight daemon is spawned DETACHED (it outlives the
// backend) yet runs off venv\Scripts\pythonw.exe, keeping venv files
// mapped past the backend teardown (#75477/#75478). Narrowly scoped
// (venv-holder-select) — external holders are never killed here.
killHermesOwnedVenvDaemons(updateRoot)
const shim = venvHermesShimPath(updateRoot)
const gate = await waitForBackendRelease(
@@ -3756,6 +3892,12 @@ async function applyUpdates(opts: { stopSafeBlockers?: boolean } = {}) {
emitUpdateProgress({ stage: 'error', message, percent: null })
startHermes().catch(() => {})
if (IS_WINDOWS) {
// The pre-gate `gateway stop --all` (#70337) took every profile's
// gateway down for an update that never happened — bring them back.
startGatewaysAfterUpdateAbort(venvHermesShimPath(updateRoot))
}
return { ok: false, error: message }
}
@@ -3805,16 +3947,21 @@ async function applyUpdates(opts: { stopSafeBlockers?: boolean } = {}) {
rememberLog(`[updates] venv-blocked: ${scanOutcome.result.processes.length} process(es) hold the install`)
emitUpdateProgress({ stage: 'error', message, percent: null })
startHermes().catch(() => {})
// Restore the gateways the pre-gate stop took down (#70337 drain
// semantics): the update aborted, so nothing else will relaunch them.
startGatewaysAfterUpdateAbort(venvHermesShimPath(updateRoot))
return { ok: false, error: 'venv-blocked', message, blockers: scanOutcome.result.processes }
}
if (scanOutcome.kind === 'probe-failure') {
const message = formatProbeFailedMessage()
const message = formatProbeFailedMessage(scanOutcome.error)
rememberLog(`[updates] venv-blocker probe failed: ${scanOutcome.error}`)
emitUpdateProgress({ stage: 'error', message, percent: null })
startHermes().catch(() => {})
// Same drain-semantics restore as the venv-blocked abort above.
startGatewaysAfterUpdateAbort(venvHermesShimPath(updateRoot))
return { ok: false, error: 'venv-probe-failed', message }
}
@@ -3943,6 +4090,11 @@ async function applyUpdates(opts: { stopSafeBlockers?: boolean } = {}) {
emitUpdateProgress({ stage: 'error', message, percent: null })
startHermes().catch(() => {})
if (IS_WINDOWS) {
// Same drain-semantics restore as the earlier abort paths (#70337).
startGatewaysAfterUpdateAbort(venvHermesShimPath(updateRoot))
}
return { ok: false, error: 'updater-spawn-failed', message }
}
@@ -5853,7 +6005,7 @@ async function saveImageFromUrl(rawUrl) {
return true
}
async function writeComposerImage(buffer, ext = '.png') {
async function writeComposerImage(buffer, ext = '.png', name = '') {
const rawExt = String(ext || '.png')
.trim()
.toLowerCase()
@@ -5864,7 +6016,19 @@ async function writeComposerImage(buffer, ext = '.png') {
await fs.promises.mkdir(dir, { recursive: true })
const stamp = new Date().toISOString().replace(/[:.]/g, '-').replace('T', '_').replace('Z', '')
const random = crypto.randomBytes(3).toString('hex')
const filePath = path.join(dir, `composer_${stamp}_${random}${safeExt}`)
const baseName = String(name || '')
.split(/[\\/]/)
.pop()
?.replace(/\.[^.]+$/, '')
const safeName = (baseName || '')
.replace(/[^\p{L}\p{N}._-]+/gu, '_')
.replace(/^[._-]+|[._-]+$/g, '')
.slice(0, 80)
const fileName = safeName ? `${safeName}_${random}${safeExt}` : `composer_${stamp}_${random}${safeExt}`
const filePath = path.join(dir, fileName)
await fs.promises.writeFile(filePath, buffer)
return filePath
@@ -9315,6 +9479,68 @@ function writeActiveDesktopProfile(name) {
return value || null
}
// True when the given pid belongs to a running process whose command line
// contains "hermes", avoiding false positives from stale gateway.pid files
// whose PID was recycled by the OS to an unrelated process.
function isHermesProcess(pid) {
try {
process.kill(pid, 0) // signal 0 = existence check, no signal sent
} catch {
return false
}
// On macOS / Linux, check the command line to avoid PID recycling false positives.
try {
const cmdline = fs.readFileSync(`/proc/${pid}/cmdline`, 'utf8')
return cmdline.includes('hermes')
} catch {
// /proc not available (macOS) — fall back to ps. Use -o args= to inspect
// the full command line, not just the process name. -o comm= would return
// "python3" for any Python process, creating false positives.
try {
const { execSync } = require('child_process')
const out = execSync(`ps -p ${pid} -o args=`, { encoding: 'utf8', timeout: 2000 })
return out.includes('hermes')
} catch {
return false
}
}
}
// Seed active-profile.json from the best available signal when the file does
// not yet exist. Runs exactly once (no-op once the file exists). Priority:
// 1. Legacy ~/.hermes/active_profile (explicit CLI choice via hermes profile use)
// 2. Running gateway (gateway.pid with verified liveness + hermes identity)
// 3. state.db heuristics (hybrid recency×size score picks the primary workspace)
// The stored JSON includes _migrated:true so the renderer can optionally surface
// a one-time notification that the profile was auto-detected.
//
// Decision logic lives in profile-migration.ts (pure + unit-tested). This wrapper
// just wires Electron/Node fs into a MigrationDeps bag and delegates.
function migrateActiveProfileIfMissing() {
migrateActiveProfileIfMissingPure(DESKTOP_PROFILE_CONFIG_PATH, {
legacyActivePath: path.join(HERMES_HOME, 'active_profile'),
hermesHome: HERMES_HOME,
profilesRoot: path.join(HERMES_HOME, 'profiles'),
existsSync: p => fs.existsSync(p),
readFileSync: (p, enc) => fs.readFileSync(p, enc),
statSync: p => fs.statSync(p),
readdirSync: (p, opts) => fs.readdirSync(p, opts as { withFileTypes: true }),
isHermesProcess,
now: () => Date.now(),
writeJson: (target, decision) => {
// Mirror writeActiveDesktopProfile's atomic-write + parent-dir-create
// semantics so the migration produces a file indistinguishable from a
// user-driven profile switch.
fs.mkdirSync(path.dirname(target), { recursive: true })
writeFileAtomic(target, JSON.stringify(decision, null, 2))
},
isValidProfileName: p => PROFILE_NAME_RE.test(p)
})
}
// Sanitize a connection config into the renderer-facing shape. With no
// `profile` this describes the global/default connection (the existing
// behavior); with a `profile` it describes that profile's per-profile remote
@@ -9851,6 +10077,7 @@ function clearManagedSshRecovery(connectionId, correlationId) {
const sshBootstrapCoordinator = createBootstrapCoordinator()
let sshQuitTeardownDone = false
let sshQuitTeardownPromise: Promise<void> | null = null
let backendQuitTeardownDone = false
function sshScopeKey(profile) {
@@ -11151,7 +11378,8 @@ async function ensureRegistryBackend(connectionId, profile, managedUpdateCorrela
return {
...primaryDescriptor,
profile: profileKey,
connectionId: id
connectionId: id,
sharedRemote: true
}
}
}
@@ -12074,6 +12302,21 @@ async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; po
// the claim, and would miss anything printed before it.
const outputTail = createBackendOutputTail()
outputTail.attach(child)
// Start watching for the READY announcement BEFORE any await (#60323):
// stdout is already flowing into the tail, and Node streams never replay
// consumed chunks to late listeners — a sentinel printed while
// claimBackendChild runs would otherwise be lost forever, timing out a
// healthy backend. The tail-buffer accessor covers any residual gap.
const portAnnouncement = waitForDashboardPortAnnouncement(child, {
bufferedOutput: () => outputTail.text(),
describeOutputTail: () => outputTail.describe(),
readyFile
})
// Mark handled so an early rejection (child dies during the claim) can't
// surface as an unhandled rejection before the Promise.race below attaches.
portAnnouncement.catch(() => {})
await claimBackendChild(child, `${backend.command} ${backend.args.join(' ')}`, profile, backendNonce, outputTail)
child.stdout.on('data', rememberLog)
@@ -12107,10 +12350,7 @@ async function spawnPoolBackend(profile, entry, opts: { forceLocal?: boolean; po
})
// Discover the ephemeral port the child bound to
const port = await Promise.race([
waitForDashboardPortAnnouncement(child, { describeOutputTail: () => outputTail.describe(), readyFile }),
startFailed
])
const port = await Promise.race([portAnnouncement, startFailed])
if (readyFile) {
fs.unlink(readyFile, () => {})
@@ -12298,6 +12538,15 @@ async function startHermes() {
return existingConnectionPromise
}
// Seed active-profile.json from legacy signals BEFORE the first
// profile-dependent read (`primaryBackendIsRemote()` on the next line, then
// `primaryProfileKey()` inside the connection IIFE below). Without this,
// remote-mode users whose preference file is missing (first boot after
// update) resolve primaryProfileKey() to 'default' inside the IIFE, then
// the remote branch returns and never runs the migration. Runs once;
// no-op when the preference file already exists.
migrateActiveProfileIfMissing()
const connectionAttempt = backendConnectionState.startAttempt()
const primaryProfile = primaryProfileKey()
@@ -12461,6 +12710,24 @@ async function startHermes() {
// later, after the claim, and would miss anything printed before it.
const primaryOutputTail = createBackendOutputTail()
primaryOutputTail.attach(hermesProcess)
// Start watching for the READY announcement BEFORE any await (#60323):
// claimBackendChild can take seconds (its Windows Get-Process probe cold
// start alone runs 2-8s) and advanceBootProgress awaits renderer IPC.
// stdout is already flowing into the tail, and Node streams never replay
// consumed chunks to late listeners, so a sentinel printed during that
// window was lost forever — the wait then hit its 90s timeout and a
// healthy backend was killed (deterministic on Windows, racy on
// macOS/Linux). The tail-buffer accessor covers any residual gap.
const portAnnouncement = waitForDashboardPortAnnouncement(hermesProcess, {
bufferedOutput: () => primaryOutputTail.text(),
describeOutputTail: () => primaryOutputTail.describe(),
readyFile
})
// Mark handled so an early rejection (child dies during the claim) can't
// surface as an unhandled rejection before the Promise.race below attaches.
portAnnouncement.catch(() => {})
await claimBackendChild(
hermesProcess,
`${backend.command} ${backend.args.join(' ')}`,
@@ -12547,13 +12814,7 @@ async function startHermes() {
await advanceBootProgress('backend.port', 'Waiting for Hermes backend to launch', 86)
// Discover the ephemeral port the child bound to
const port = await Promise.race([
waitForDashboardPortAnnouncement(hermesProcess, {
describeOutputTail: () => primaryOutputTail.describe(),
readyFile
}),
backendStartFailed
])
const port = await Promise.race([portAnnouncement, backendStartFailed])
if (readyFile) {
fs.unlink(readyFile, () => {})
@@ -14480,6 +14741,21 @@ const hudIpc = registerHudIpc({
}
})
ipcMain.handle('hermes:backend:recycle', async (_event, profile) => {
// Models-page recovery after a code-skew 503 (#97046): kill the owned
// SSH serve (if any) before the local child so reconnect cannot reuse a
// stale lockfile. Soft primary teardown keeps the renderer shell mounted.
await recycleOwnedBackend({
notifyApplied: sendConnectionApplied,
primaryProfile: primaryProfileKey(),
profile: typeof profile === 'string' ? profile : '',
teardownPool: teardownPoolBackendAndWait,
teardownPrimary: () => teardownPrimaryBackendAndWait({ soft: true }),
teardownSsh: value => teardownSshConnection(value || null)
})
return { ok: true }
})
ipcMain.handle('hermes:bootstrap:reset', async () => {
// Renderer's "Reload and retry" path. Clear the latched failure and
// reset connection state so the next startHermes() call restarts the
@@ -16375,6 +16651,12 @@ ipcMain.handle('hermes:context-menu:guest-add-word', (_event, payload) => {
}
})
ipcMain.handle('hermes:capturePreview', async (_event, payload) => {
const guest = electronWebContents.fromId(Number(payload?.webContentsId))
return capturePreviewContents(guest, payload?.rect, payload?.viewport)
})
ipcMain.handle('hermes:saveImageBuffer', async (_event, payload) => {
const data = payload?.data
@@ -16384,7 +16666,7 @@ ipcMain.handle('hermes:saveImageBuffer', async (_event, payload) => {
const buffer = Buffer.isBuffer(data) ? data : Buffer.from(data)
return writeComposerImage(buffer, payload?.ext || '.png')
return writeComposerImage(buffer, payload?.ext || '.png', payload?.name)
})
ipcMain.handle('hermes:saveClipboardImage', async () => {
@@ -16521,6 +16803,15 @@ ipcMain.on('hermes:translucency:support', event => {
event.returnValue = { glass: GLASS_SUPPORTED, translucency: TRANSLUCENCY_SUPPORTED }
})
// Launch-flag facts the renderer needs before first paint (same sendSync
// pattern as translucency). `--local` gates every local-models GUI surface;
// it arrives from `hermes desktop --local` or directly on Hermes.exe (a
// shortcut edit), and survives self-relaunches because collectRelaunchArgs
// only strips internal flags.
ipcMain.on('hermes:launch-flags', event => {
event.returnValue = { localModels: process.argv.includes('--local') }
})
ipcMain.on('hermes:translucency', (_event, payload) => {
const next = normalizeTranslucency(payload, GLASS_SUPPORTED)
const previous = translucencyState
@@ -16950,10 +17241,26 @@ ipcMain.handle('hermes:version', async () => {
platform: process.platform,
hermesRoot: resolveUpdateRoot(),
bundleOutOfSync: skew.outOfSync,
bundleCommitsBehind: skew.desktopCommitsBehind
bundleCommitsBehind: skew.desktopCommitsBehind,
// True when the bundle on disk is not the one this process loaded — a
// plain app restart (no rebuild, no installer) clears the skew above.
// Packaged only: a dev `--build-only` rewrites build/install-stamp.json
// under a running `npm start`, which is a rebuild the developer asked for,
// not a torn install to offer a restart for.
bundleSwapPending: IS_PACKAGED && detectBundleSwap(INSTALL_STAMP, loadInstallStamp())
}
})
// The About page's "Restart Hermes" button (shown when bundleSwapPending):
// load the already-swapped bundle without asking the user to quit manually.
// app.relaunch() re-executes by path, so the fresh process picks up whatever
// bundle now lives there.
ipcMain.handle('hermes:app:relaunch', async () => {
rememberLog('[updates] renderer requested an app relaunch (swapped bundle pending)')
app.relaunch({ args: buildNoSandboxRelaunchArgs(process.argv.slice(1)) })
void exitAfterBackendShutdown(0)
})
// ===========================================================================
// Uninstall — remove the Chat GUI (and optionally the agent / user data).
// ===========================================================================
@@ -17539,20 +17846,39 @@ app.on('before-quit', event => {
})
}
if ((sshConnections.size > 0 || sshBootstrapCoordinator.promises().length > 0) && !sshQuitTeardownDone) {
// backendShutdown.finally() re-enters before-quit. teardownSshConnection
// already deleted the map entries, so size===0 would skip the kill wait
// and let window-X quit finish while SSH exec is still running (#91668).
if (
sshQuitShouldBlock({
teardownDone: sshQuitTeardownDone,
connectionCount: sshConnections.size,
bootstrapPending: sshBootstrapCoordinator.promises().length,
inFlight: sshQuitTeardownPromise
})
) {
event.preventDefault()
const scopes = [...sshConnections.keys()]
const pending = Promise.allSettled([
...scopes.map(scope => teardownSshConnection(scope || null)),
...sshBootstrapCoordinator.promises()
])
if (!sshQuitTeardownPromise) {
const scopes = [...sshConnections.keys()]
// cleanupStale waits up to 5s for the owned pid to exit (50 * 100ms).
// The previous 4s race could close SSH first and leave serve --isolated
// reparented to pid 1.
void Promise.race([pending, new Promise(resolve => setTimeout(resolve, 6_000))]).then(async () => {
await sshBootstrapCoordinator.forceCleanupAll()
const pending = Promise.allSettled([
...scopes.map(scope => teardownSshConnection(scope || null)),
...sshBootstrapCoordinator.promises()
])
// cleanupStale waits up to 5s for the owned pid to exit (50 * 100ms).
// The previous 4s race could close SSH first and leave serve --isolated
// reparented to pid 1. Latch this promise BEFORE those deletes land so
// a re-entrant quit still waits.
sshQuitTeardownPromise = Promise.race([pending, new Promise<void>(resolve => setTimeout(resolve, 6_000))]).then(
async () => {
await sshBootstrapCoordinator.forceCleanupAll()
}
)
}
void sshQuitTeardownPromise.then(() => {
sshQuitTeardownDone = true
app.quit()
})
+9 -1
View File
@@ -10,10 +10,14 @@ import { contextBridge, ipcRenderer, webFrame, webUtils } from 'electron'
const translucencySupport = ipcRenderer.sendSync('hermes:translucency:support')
const hudWindowing = ipcRenderer.sendSync('hermes:hud:windowing')
const hudNativeDrag = hudWindowing?.nativeDrag === true
const launchFlags = ipcRenderer.sendSync('hermes:launch-flags')
contextBridge.exposeInMainWorld('hermesDesktop', {
glassSupported: translucencySupport?.glass === true,
translucencySupported: translucencySupport?.translucency === true,
// Launch-flag fact: the app was started with --local, so the renderer may
// show the local-models surfaces. Static for the window's lifetime.
localModelsEnabled: launchFlags?.localModels === true,
getConnection: profile => ipcRenderer.invoke('hermes:connection', profile),
// Registry-scoped backend resolution: { connectionId, profile } → descriptor.
getConnectionFor: payload => ipcRenderer.invoke('hermes:connection:for', payload),
@@ -249,7 +253,8 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
return () => ipcRenderer.removeListener('hermes:context-menu-spellcheck', listener)
},
saveImageBuffer: (data, ext) => ipcRenderer.invoke('hermes:saveImageBuffer', { data, ext }),
saveImageBuffer: (data, ext, name) => ipcRenderer.invoke('hermes:saveImageBuffer', { data, ext, name }),
capturePreview: payload => ipcRenderer.invoke('hermes:capturePreview', payload),
saveClipboardImage: () => ipcRenderer.invoke('hermes:saveClipboardImage'),
getPathForFile: file => {
try {
@@ -350,6 +355,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
}
},
terminal: {
attach: id => ipcRenderer.invoke('hermes:terminal:attach', id),
cwd: id => ipcRenderer.invoke('hermes:terminal:cwd', id),
dispose: id => ipcRenderer.invoke('hermes:terminal:dispose', id),
resize: (id, size) => ipcRenderer.invoke('hermes:terminal:resize', id, size),
@@ -474,6 +480,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
// reload mid-bootstrap.
getBootstrapState: () => ipcRenderer.invoke('hermes:bootstrap:get'),
continueBootstrapLocal: () => ipcRenderer.invoke('hermes:bootstrap:continue-local'),
recycleBackend: profile => ipcRenderer.invoke('hermes:backend:recycle', profile),
resetBootstrap: () => ipcRenderer.invoke('hermes:bootstrap:reset'),
repairBootstrap: () => ipcRenderer.invoke('hermes:bootstrap:repair'),
cancelBootstrap: () => ipcRenderer.invoke('hermes:bootstrap:cancel'),
@@ -484,6 +491,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', {
return () => ipcRenderer.removeListener('hermes:bootstrap:event', listener)
},
getVersion: () => ipcRenderer.invoke('hermes:version'),
relaunchApp: () => ipcRenderer.invoke('hermes:app:relaunch'),
getRemoteDisplayReason: () => ipcRenderer.invoke('hermes:get-remote-display-reason'),
uninstall: {
summary: () => ipcRenderer.invoke('hermes:uninstall:summary'),
@@ -0,0 +1,97 @@
import assert from 'node:assert/strict'
import { test } from 'vitest'
import { capturePreviewContents, mapViewportRectToImage, normalizeCaptureRect } from './preview-capture'
test('normalizeCaptureRect floors origin and ceils size, never zero', () => {
assert.deepEqual(normalizeCaptureRect({ height: 10.2, width: 0.4, x: -3.2, y: 1.8 }), {
height: 11,
width: 1,
x: 0,
y: 1
})
})
test('mapViewportRectToImage scales CSS pixels onto a DPR bitmap and clamps', () => {
assert.deepEqual(
mapViewportRectToImage(
{ height: 20, width: 40, x: 10, y: 8 },
{ height: 100, width: 200 },
{ height: 200, width: 400 }
),
{ height: 40, width: 80, x: 20, y: 16 }
)
assert.equal(
mapViewportRectToImage(
{ height: 20, width: 40, x: 500, y: 8 },
{ height: 100, width: 200 },
{ height: 200, width: 400 }
),
null
)
})
test('capturePreviewContents captures the visible page then crops in bitmap space', async () => {
const png = Buffer.from([0x89, 0x50, 0x4e, 0x47])
const dataUrl = await capturePreviewContents(
{
capturePage: async rect => {
assert.equal(rect, undefined)
return {
crop: mapped => {
assert.deepEqual(mapped, { height: 40, width: 80, x: 4, y: 8 })
return { isEmpty: () => false, toPNG: () => png }
},
getSize: () => ({ height: 200, width: 400 }),
isEmpty: () => false,
toPNG: () => {
throw new Error('should crop first')
}
}
},
isDestroyed: () => false
},
{ height: 20, width: 40, x: 2, y: 4 },
{ height: 100, width: 200 }
)
assert.equal(dataUrl, `data:image/png;base64,${png.toString('base64')}`)
})
test('capturePreviewContents keeps the visible page when the CSS rect misses the bitmap', async () => {
const png = Buffer.from([0x89, 0x50, 0x4e, 0x47])
const dataUrl = await capturePreviewContents(
{
capturePage: async () => ({
crop: () => {
throw new Error('should not crop an off-screen rect')
},
getSize: () => ({ height: 200, width: 400 }),
isEmpty: () => false,
toPNG: () => png
}),
isDestroyed: () => false
},
{ height: 20, width: 40, x: 10, y: 800 },
{ height: 100, width: 200 }
)
assert.equal(dataUrl, `data:image/png;base64,${png.toString('base64')}`)
})
test('capturePreviewContents fails closed on a destroyed guest', async () => {
await assert.rejects(
capturePreviewContents({
capturePage: async () => {
throw new Error('should not run')
},
isDestroyed: () => true
}),
/gone/
)
})
+131
View File
@@ -0,0 +1,131 @@
/**
* Crop a preview webview via webContents.capturePage.
*
* Electron's rect argument is unreliable on guest webviews (empty NativeImage
* on Windows, DPI-shifted crops). Capture the visible viewport, then crop in
* bitmap space from CSS viewport coordinates.
*
* Missing/destroyed guests fail closed — never a blank PNG that would look
* like a successful shot of nothing.
*/
export interface CaptureRect {
height: number
width: number
x: number
y: number
}
export interface CaptureViewport {
height: number
width: number
}
export interface CaptureGuest {
capturePage: (rect?: CaptureRect) => Promise<CaptureImage>
isDestroyed: () => boolean
}
export interface CaptureImage {
crop?: (rect: CaptureRect) => CaptureImage
getSize?: () => CaptureViewport
isEmpty: () => boolean
toPNG: () => Buffer
}
export function normalizeCaptureRect(rect?: CaptureRect): CaptureRect | undefined {
if (!rect) {
return undefined
}
return {
height: Math.max(1, Math.ceil(rect.height)),
width: Math.max(1, Math.ceil(rect.width)),
x: Math.max(0, Math.floor(rect.x)),
y: Math.max(0, Math.floor(rect.y))
}
}
/** Map a CSS-pixel viewport rect onto a (possibly DPR-scaled) bitmap. */
export function mapViewportRectToImage(
rect: CaptureRect,
viewport: CaptureViewport,
image: CaptureViewport
): CaptureRect | null {
const viewW = Math.max(1, viewport.width)
const viewH = Math.max(1, viewport.height)
const scaleX = image.width / viewW
const scaleY = image.height / viewH
const left = rect.x * scaleX
const top = rect.y * scaleY
const right = (rect.x + rect.width) * scaleX
const bottom = (rect.y + rect.height) * scaleY
const x = Math.max(0, Math.floor(left))
const y = Math.max(0, Math.floor(top))
const maxX = Math.min(image.width, Math.ceil(right))
const maxY = Math.min(image.height, Math.ceil(bottom))
const width = maxX - x
const height = maxY - y
if (width < 1 || height < 1) {
return null
}
return { height, width, x, y }
}
function toDataUrl(image: CaptureImage): string {
if (!image || image.isEmpty()) {
throw new Error('preview capture was empty')
}
return `data:image/png;base64,${image.toPNG().toString('base64')}`
}
export async function capturePreviewContents(
guest: CaptureGuest | null | undefined,
rect?: CaptureRect,
viewport?: CaptureViewport
): Promise<string> {
if (!guest || guest.isDestroyed()) {
throw new Error('preview guest is gone')
}
const image = await guest.capturePage()
if (!image || image.isEmpty()) {
throw new Error('preview capture was empty')
}
const crop = normalizeCaptureRect(rect)
if (!crop) {
return toDataUrl(image)
}
const size = image.getSize?.()
const view = viewport && viewport.width > 0 && viewport.height > 0 ? viewport : size
if (size && view && typeof image.crop === 'function') {
const mapped = mapViewportRectToImage(crop, view, size)
if (mapped) {
const cropped = image.crop(mapped)
if (cropped && !cropped.isEmpty()) {
return toDataUrl(cropped)
}
}
// Off-screen CSS rects (document Y after scroll, DPI mismatch) used to
// throw here and the composer kept the stale pick-time crop. The visible
// page is a better fallback than the wrong slice of the article.
return toDataUrl(image)
}
try {
return toDataUrl(await guest.capturePage(crop))
} catch {
return toDataUrl(image)
}
}
@@ -0,0 +1,687 @@
/**
* Tests for electron/profile-migration.ts — pure migration-decision helpers for
* the active-profile.json first-boot seeding.
*
* Run with: `vitest run` (wired via the `electronNative` project in
* `apps/desktop/vitest.config.ts`, which discovers tests under `electron/`).
*
* These tests cover the four axes the maintainer explicitly required:
* 1. precedence — legacy > single-running-gateway > state.db heuristic
* 2. stale PID rejection — recycled PID must not pass as a running gateway
* 3. fallback behavior — single-profile installs and missing files are no-ops
* 4. remote-boot path — verified by code review (the migration is moved to the
* top of startHermes() in main.ts, before primaryProfileKey() is read); the
* pure decision logic that the function relies on is covered below.
*/
import assert from 'node:assert/strict'
import type { Dirent } from 'node:fs'
import { test } from 'vitest'
import {
decideMigration,
findRunningGatewayProfiles,
listProfileDirs,
migrateActiveProfileIfMissing,
PROFILE_SCORE_MIN_SIZE_BYTES,
profileGatewayPidPath,
profileStateDbPath,
readExistingPreference,
readLegacyActiveProfile,
scoreStateDb,
withDefaultCandidate
} from './profile-migration'
// ---------------------------------------------------------------------------
// Fixtures
// ---------------------------------------------------------------------------
// Mirrors the production PROFILE_NAME_RE in main.ts (regex matches "default";
// callers like readLegacyActiveProfile must reject "default" explicitly).
const PROFILE_NAME_RE = /^[a-z0-9][a-z0-9_-]{0,63}$/
// Production-shaped validator: regex matches, but 'default' is excluded because
// it's the implicit fallback, never a user-chosen CLI value.
const isValidProfileName = (n: string) => n !== 'default' && PROFILE_NAME_RE.test(n)
const NOW = 1_700_000_000_000
type FileFixture = { content?: string; size?: number; mtime?: number; dir?: boolean }
function makeFs(files: Record<string, FileFixture>) {
const map = new Map(Object.entries(files))
const existsSync = (p: string) => map.has(p)
const readFileSync = (p: string, _enc: 'utf8') => {
const e = map.get(p)
if (!e || e.content === undefined) {
throw new Error(`ENOENT: ${p}`)
}
return e.content
}
const statSync = (p: string) => {
const e = map.get(p)
if (!e || e.size === undefined) {
throw new Error(`ENOENT: ${p}`)
}
return { size: e.size, mtimeMs: e.mtime ?? NOW - 86_400_000 }
}
const readdirSync = (p: string, _options?: { withFileTypes?: boolean }): Dirent[] => {
const names = new Set<string>()
// 1. Explicit dir markers at this exact level.
for (const [fullPath, fixture] of map.entries()) {
if (fullPath === p && fixture.dir) {
// This is the dir itself; not a child. Skip.
continue
}
if (fullPath.startsWith(p + '/')) {
const rest = fullPath.slice(p.length + 1)
if (!rest.includes('/') && fixture.dir) {
names.add(rest)
}
}
}
// 2. Implicit directories: any nested path under `p` implies the intermediate
// directory exists (mirrors how mkdir({recursive:true}) populates the tree).
for (const key of map.keys()) {
if (!key.startsWith(p + '/')) {
continue
}
const rest = key.slice(p.length + 1)
const parts = rest.split('/')
if (parts.length < 2) {
continue
}
names.add(parts[0])
}
const entries: Dirent[] = []
for (const name of names) {
entries.push({
name,
isDirectory: () => true,
isFile: () => false,
isBlockDevice: () => false,
isCharacterDevice: () => false,
isSymbolicLink: () => false,
isFIFO: () => false,
isSocket: () => false
} as unknown as Dirent)
}
if (entries.length === 0 && !map.has(p)) {
throw new Error(`ENOENT: ${p}`)
}
return entries
}
return { existsSync, readFileSync, statSync, readdirSync }
}
function baseDeps(overrides: Record<string, unknown> = {}) {
const fs = makeFs({})
return {
legacyActivePath: '/home/u/.hermes/active_profile',
hermesHome: '/home/u/.hermes',
profilesRoot: '/home/u/.hermes/profiles',
existsSync: fs.existsSync,
readFileSync: fs.readFileSync,
statSync: fs.statSync,
readdirSync: fs.readdirSync,
isHermesProcess: () => true,
now: () => NOW,
writeJson: () => {},
isValidProfileName,
...overrides
}
}
// ---------------------------------------------------------------------------
// readLegacyActiveProfile
// ---------------------------------------------------------------------------
test('readLegacyActiveProfile returns null when file missing', () => {
assert.equal(
readLegacyActiveProfile(
'/missing',
() => {
throw new Error('ENOENT')
},
isValidProfileName
),
null
)
})
test('readLegacyActiveProfile returns trimmed name on success', () => {
assert.equal(
readLegacyActiveProfile('/p', () => ' coder \n' as unknown as string, isValidProfileName),
'coder'
)
})
test('readLegacyActiveProfile returns undefined when present but invalid (special chars)', () => {
assert.equal(
readLegacyActiveProfile('/p', () => 'BAD NAME!' as unknown as string, isValidProfileName),
undefined
)
})
test('readLegacyActiveProfile rejects default as a legacy CLI choice', () => {
// "default" matches PROFILE_NAME_RE but is implicit; the explicit guard
// suppresses it so the heuristic rung still fires.
assert.equal(
readLegacyActiveProfile('/p', () => 'default' as unknown as string, isValidProfileName),
undefined
)
})
test('readLegacyActiveProfile returns null for empty/whitespace content', () => {
assert.equal(
readLegacyActiveProfile('/p', () => ' \n' as unknown as string, isValidProfileName),
null
)
})
// ---------------------------------------------------------------------------
// findRunningGatewayProfiles
// ---------------------------------------------------------------------------
test('findRunningGatewayProfiles returns [] when no pid files exist', () => {
const fs = makeFs({ '/home/u/.hermes/profiles/coder': { dir: true } })
assert.deepEqual(
findRunningGatewayProfiles('/home/u/.hermes/profiles', ['coder'], {
...fs,
isHermesProcess: () => true
}),
[]
)
})
test('findRunningGatewayProfiles drops stale (non-hermes) recycled PIDs', () => {
// Two profiles have pid files, but only coder's pid is a live hermes process.
// The recycled PID at 5678 belongs to an unrelated process (e.g. Chrome).
const fs = makeFs({
'/home/u/.hermes/profiles/coder/gateway.pid': { content: '{"pid":1234}' },
'/home/u/.hermes/profiles/writer/gateway.pid': { content: '{"pid":5678}' }
})
const deps = {
...fs,
isHermesProcess: (pid: number) => pid === 1234
}
assert.deepEqual(findRunningGatewayProfiles('/home/u/.hermes/profiles', ['coder', 'writer'], deps), ['coder'])
})
test('findRunningGatewayProfiles tolerates malformed pid files', () => {
const fs = makeFs({
'/home/u/.hermes/profiles/coder/gateway.pid': { content: '{not json' }
})
assert.deepEqual(
findRunningGatewayProfiles('/home/u/.hermes/profiles', ['coder'], {
...fs,
isHermesProcess: () => true
}),
[]
)
})
test('findRunningGatewayProfiles drops non-integer and non-positive PIDs', () => {
const fs = makeFs({
'/home/u/.hermes/profiles/coder/gateway.pid': { content: '{"pid":-1}' },
'/home/u/.hermes/profiles/writer/gateway.pid': { content: '{"pid":1.5}' },
'/home/u/.hermes/profiles/extra/gateway.pid': { content: '{"pid":0}' }
})
assert.deepEqual(
findRunningGatewayProfiles('/home/u/.hermes/profiles', ['coder', 'writer', 'extra'], {
...fs,
isHermesProcess: () => true
}),
[]
)
})
test('findRunningGatewayProfiles preserves order of allProfiles', () => {
const fs = makeFs({
'/home/u/.hermes/profiles/coder/gateway.pid': { content: '{"pid":1}' },
'/home/u/.hermes/profiles/writer/gateway.pid': { content: '{"pid":2}' }
})
const deps = { ...fs, isHermesProcess: () => true }
assert.deepEqual(findRunningGatewayProfiles('/home/u/.hermes/profiles', ['coder', 'writer'], deps), [
'coder',
'writer'
])
})
// ---------------------------------------------------------------------------
// scoreStateDb
// ---------------------------------------------------------------------------
test('scoreStateDb returns null on missing file', () => {
assert.equal(
scoreStateDb('/missing', NOW, () => {
throw new Error('ENOENT')
}),
null
)
})
test('scoreStateDb floors recency weight at 0.1 for ancient files', () => {
const ancient = scoreStateDb('/p', NOW, () => ({ size: 10 * 1024 * 1024, mtimeMs: 0 }))
const fresh = scoreStateDb('/p', NOW, () => ({ size: 10 * 1024 * 1024, mtimeMs: NOW - 86_400_000 }))
// Ancient still scores > 0 because recency is floored at 0.1.
assert.ok(ancient !== null && ancient > 0)
assert.ok(fresh !== null && fresh > ancient!)
})
test('scoreStateDb prefers larger DB at similar recency', () => {
// 409 MB primary workspace vs 28 MB secondary — primary wins even with a
// slightly newer mtime on the secondary.
const big = scoreStateDb('/big', NOW, () => ({ size: 409 * 1024 * 1024, mtimeMs: NOW - 86_400_000 }))
const small = scoreStateDb('/small', NOW, () => ({ size: 28 * 1024 * 1024, mtimeMs: NOW - 60_000 }))
assert.ok(big !== null && small !== null && big > small)
})
test('scoreStateDb floors size weight at log10(MIN_SIZE) for tiny files', () => {
// 100 bytes and PROFILE_SCORE_MIN_SIZE_BYTES score the same on the size axis
// (recency is the same here, so the score is identical).
const tiny = scoreStateDb('/tiny', NOW, () => ({ size: 100, mtimeMs: NOW - 86_400_000 }))
const min = scoreStateDb('/min', NOW, () => ({ size: PROFILE_SCORE_MIN_SIZE_BYTES, mtimeMs: NOW - 86_400_000 }))
assert.equal(tiny, min)
})
// ---------------------------------------------------------------------------
// listProfileDirs
// ---------------------------------------------------------------------------
test('listProfileDirs returns [] for missing profiles root', () => {
const fs = makeFs({})
assert.deepEqual(listProfileDirs(baseDeps({ ...fs })), [])
})
test('listProfileDirs includes default and any name passing the regex', () => {
const fs = makeFs({
'/home/u/.hermes/profiles/default': { dir: true },
'/home/u/.hermes/profiles/coder': { dir: true },
'/home/u/.hermes/profiles/writer': { dir: true }
})
assert.deepEqual(listProfileDirs(baseDeps({ ...fs })), ['default', 'coder', 'writer'])
})
test('listProfileDirs skips files (not directories) and invalid names', () => {
const fs = makeFs({
'/home/u/.hermes/profiles/default': { dir: true },
'/home/u/.hermes/profiles/some-file.txt': { dir: false },
'/home/u/.hermes/profiles/UPPERCASE': { dir: true } // regex rejects
})
assert.deepEqual(listProfileDirs(baseDeps({ ...fs })), ['default'])
})
// ---------------------------------------------------------------------------
// decideMigration
// ---------------------------------------------------------------------------
test('decideMigration prefers legacy over everything else', () => {
const deps = baseDeps()
const d = decideMigration('coder', ['writer'], ['coder', 'writer'], deps, () => null)
assert.deepEqual(d, { profile: 'coder' })
assert.equal(d?._migrated, undefined)
})
test('decideMigration prefers a single running gateway profile', () => {
const deps = baseDeps()
const d = decideMigration(null, ['coder'], ['coder', 'writer'], deps, () => null)
assert.deepEqual(d, { profile: 'coder' })
assert.equal(d?._migrated, undefined)
})
test('decideMigration falls through to scoring when multiple gateways run', () => {
// When running.length > 1 the heuristic must score the running set, not all
// profiles, so a stopped profile can't win against two live ones.
const deps = baseDeps()
const d = decideMigration(null, ['coder', 'writer'], ['coder', 'writer'], deps, p =>
p.endsWith('/writer/state.db') ? 50 : 10
)
assert.deepEqual(d, { profile: 'writer', _migrated: true })
})
test('decideMigration returns null when no candidate scores and legacy is invalid', () => {
const deps = baseDeps()
assert.equal(
decideMigration(undefined, [], ['coder', 'writer'], deps, () => null),
null
)
})
test('decideMigration suppresses write when best is default (single-profile fallback)', () => {
// The whole point of the migration is to migrate AWAY from default when a
// better candidate exists. If 'default' wins the score, the install is
// default-primary and we leave it alone. Default's DB is $HERMES_HOME/state.db,
// not profiles/default/state.db.
const deps = baseDeps()
const d = decideMigration(null, [], ['default', 'coder'], deps, p => (p.endsWith('/.hermes/state.db') ? 99 : 50))
assert.equal(d, null)
})
test('decideMigration still flags _migrated when legacy is invalid (undefined) but a heuristic wins', () => {
// legacy === undefined means "file was present but malformed" — fall through to
// the heuristic and mark as migrated.
const deps = baseDeps()
const d = decideMigration(undefined, [], ['coder'], deps, () => 10)
assert.deepEqual(d, { profile: 'coder', _migrated: true })
})
// ---------------------------------------------------------------------------
// migrateActiveProfileIfMissing (orchestrator)
// ---------------------------------------------------------------------------
test('migrateActiveProfileIfMissing is a no-op when a user-selected preference file exists', () => {
// No `_migrated` flag = explicit user/CLI choice. Even a huge other profile
// must not steal the pin.
let written: unknown = null
const fs = makeFs({
'/cfg/active-profile.json': { content: '{"profile":"coder"}' },
'/home/u/.hermes/profiles/coder': { dir: true },
'/home/u/.hermes/profiles/writer': { dir: true },
'/home/u/.hermes/profiles/writer/state.db': { size: 400 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
test('migrateActiveProfileIfMissing writes legacy choice with no _migrated flag', () => {
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/active_profile': { content: 'coder' },
'/home/u/.hermes/profiles/coder': { dir: true },
'/home/u/.hermes/profiles/writer': { dir: true }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), true)
assert.deepEqual(written, { profile: 'coder' })
})
test('migrateActiveProfileIfMissing writes heuristic choice with _migrated=true', () => {
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/profiles/coder/state.db': { size: 50 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/writer/state.db': { size: 200 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), true)
assert.deepEqual(written, { profile: 'writer', _migrated: true })
})
test('migrateActiveProfileIfMissing is a no-op for single-profile (default-only) installs', () => {
// No heuristic candidate can beat 'default', so the orchestrator must NOT
// write a file — preserves legacy launch behavior for the 99% case.
// Production default DB is ~/.hermes/state.db, not profiles/default/state.db.
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/state.db': { size: 10 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
test('migrateActiveProfileIfMissing is a no-op when no profiles directory exists', () => {
let written: unknown = null
const fs = makeFs({})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
test('migrateActiveProfileIfMissing prefers a single running gateway over heuristics', () => {
// running=['coder'] regardless of what the heuristic would score — gateway
// ownership is the strongest signal we have for the active profile.
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/profiles/coder/gateway.pid': { content: '{"pid":42}' },
'/home/u/.hermes/profiles/writer/state.db': { size: 500 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
isHermesProcess: (pid: number) => pid === 42,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), true)
assert.deepEqual(written, { profile: 'coder' })
})
// ---------------------------------------------------------------------------
// Production layout: default is ~/.hermes, not ~/.hermes/profiles/default
// ---------------------------------------------------------------------------
test('profileStateDbPath puts default at hermesHome, named under profilesRoot', () => {
assert.equal(profileStateDbPath('default', '/home/u/.hermes', '/home/u/.hermes/profiles'), '/home/u/.hermes/state.db')
assert.equal(
profileStateDbPath('conduit', '/home/u/.hermes', '/home/u/.hermes/profiles'),
'/home/u/.hermes/profiles/conduit/state.db'
)
})
test('profileGatewayPidPath puts default at hermesHome', () => {
assert.equal(
profileGatewayPidPath('default', '/home/u/.hermes', '/home/u/.hermes/profiles'),
'/home/u/.hermes/gateway.pid'
)
assert.equal(
profileGatewayPidPath('coder', '/home/u/.hermes', '/home/u/.hermes/profiles'),
'/home/u/.hermes/profiles/coder/gateway.pid'
)
})
test('withDefaultCandidate always leads with default and dedupes', () => {
assert.deepEqual(withDefaultCandidate([]), ['default'])
assert.deepEqual(withDefaultCandidate(['conduit']), ['default', 'conduit'])
assert.deepEqual(withDefaultCandidate(['default', 'conduit']), ['default', 'conduit'])
})
test('findRunningGatewayProfiles sees default gateway.pid at hermesHome', () => {
const fs = makeFs({
'/home/u/.hermes/gateway.pid': { content: '{"pid":99}' },
'/home/u/.hermes/profiles/coder/gateway.pid': { content: '{"pid":11}' }
})
assert.deepEqual(
findRunningGatewayProfiles('/home/u/.hermes/profiles', ['default', 'coder'], {
...fs,
hermesHome: '/home/u/.hermes',
isHermesProcess: pid => pid === 99
}),
['default']
)
})
test('migrateActiveProfileIfMissing does not pin a tiny named profile over a large default DB', () => {
// Regression for #100576: first-boot after update listed only
// ~/.hermes/profiles/<name>, never scored ~/.hermes/state.db, and wrote
// { profile: named, _migrated: true }.
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/profiles/conduit': { dir: true },
'/home/u/.hermes/state.db': { size: 409 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/conduit/state.db': { size: 2 * 1024 * 1024, mtime: NOW - 60_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
test('migrateActiveProfileIfMissing still pins a named profile that actually beats default', () => {
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/profiles/work': { dir: true },
'/home/u/.hermes/state.db': { size: 5 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/work/state.db': { size: 200 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), true)
assert.deepEqual(written, { profile: 'work', _migrated: true })
})
test('migrateActiveProfileIfMissing does not pin default when only default gateway is running', () => {
let written: unknown = null
const fs = makeFs({
'/home/u/.hermes/gateway.pid': { content: '{"pid":7}' },
'/home/u/.hermes/state.db': { size: 10 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
isHermesProcess: (pid: number) => pid === 7,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
test('readExistingPreference treats _migrated as heuristic-owned', () => {
const fs = makeFs({
'/cfg/active-profile.json': { content: '{"profile":"conduit","_migrated":true}' }
})
assert.deepEqual(readExistingPreference('/cfg/active-profile.json', fs.readFileSync), {
profile: 'conduit',
migrated: true
})
})
test('migrateActiveProfileIfMissing repairs a pre-existing heuristic pin when default now wins', () => {
// Sol P1 / #100576: file already exists with _migrated:true so first-boot
// skip left affected installs stuck. Re-score and clear.
let written: unknown = null
const fs = makeFs({
'/cfg/active-profile.json': { content: '{"profile":"conduit","_migrated":true}' },
'/home/u/.hermes/profiles/conduit': { dir: true },
'/home/u/.hermes/state.db': { size: 409 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/conduit/state.db': { size: 2 * 1024 * 1024, mtime: NOW - 60_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), true)
assert.deepEqual(written, { profile: null })
})
test('migrateActiveProfileIfMissing leaves a still-correct heuristic pin alone', () => {
let written: unknown = null
const fs = makeFs({
'/cfg/active-profile.json': { content: '{"profile":"work","_migrated":true}' },
'/home/u/.hermes/profiles/work': { dir: true },
'/home/u/.hermes/state.db': { size: 5 * 1024 * 1024, mtime: NOW - 86_400_000 },
'/home/u/.hermes/profiles/work/state.db': { size: 200 * 1024 * 1024, mtime: NOW - 86_400_000 }
})
const deps = baseDeps({
...fs,
writeJson: (_p: string, payload: unknown) => {
written = payload
}
})
assert.equal(migrateActiveProfileIfMissing('/cfg/active-profile.json', deps), false)
assert.equal(written, null)
})
+311
View File
@@ -0,0 +1,311 @@
// Pure migration-decision helpers for active-profile.json first-boot seeding.
// Extracted from main.ts so they're unit-testable without Electron.
//
// The helpers here are deliberately side-effect-free except for the orchestrator
// (`migrateActiveProfileIfMissing`), which only writes the preference file when
// every fs/path concern is funnelled through the injected `MigrationDeps` bag.
// Tests inject synthetic fs maps; production wires real fs/path.
import type { Dirent } from 'node:fs'
// Floor for the size dimension of the hybrid score. A 100-byte file (effectively
// empty) and a 1 KiB file score the same on the size axis — both are too small to
// be the user's primary workspace on their own.
export const PROFILE_SCORE_MIN_SIZE_BYTES = 1024
export interface MigrationDeps {
legacyActivePath: string
/** Default profile home (`~/.hermes`). Default's state.db and gateway.pid live here. */
hermesHome: string
/** Named-profile root (`~/.hermes/profiles`). Does not contain `default`. */
profilesRoot: string
existsSync: (path: string) => boolean
readFileSync: (path: string, encoding: 'utf8') => string
statSync: (path: string) => { size: number; mtimeMs: number }
readdirSync: (path: string, options?: { withFileTypes?: boolean }) => Dirent[]
isHermesProcess: (pid: number) => boolean
now: () => number
writeJson: (path: string, payload: MigrationDecision) => void
isValidProfileName: (name: string) => boolean
}
export interface MigrationDecision {
profile: string | null
/** True when chosen from the state.db heuristic (auto-detected), undefined when explicit. */
_migrated?: boolean
}
/**
* Production layout: default IS `hermesHome`; named profiles are children of
* `profilesRoot`. There is no `profiles/default` directory on a normal install.
*/
export function profileStateDbPath(name: string, hermesHome: string, profilesRoot: string): string {
return name === 'default' ? `${hermesHome}/state.db` : `${profilesRoot}/${name}/state.db`
}
export function profileGatewayPidPath(name: string, hermesHome: string, profilesRoot: string): string {
return name === 'default' ? `${hermesHome}/gateway.pid` : `${profilesRoot}/${name}/gateway.pid`
}
function resolveHermesHome(profilesRoot: string, hermesHome?: string): string {
if (hermesHome) {
return hermesHome
}
// Tests that predate hermesHome pass only profilesRoot.
for (const suffix of ['/profiles', '\\profiles']) {
if (profilesRoot.endsWith(suffix)) {
return profilesRoot.slice(0, -suffix.length)
}
}
return profilesRoot
}
/**
* Parse the legacy CLI-sticky file. Returns the trimmed name on success, null when
* missing/unreadable/empty, undefined when present but invalid (so the caller can
* distinguish "explicitly chose default" from "no legacy file at all"). The
* `'default'` profile is always rejected here — it's an implicit fallback, never a
* user-chosen CLI value, and accepting it would suppress the heuristic rung that
* is the whole point of this migration.
*/
export function readLegacyActiveProfile(
legacyActivePath: string,
readFile: MigrationDeps['readFileSync'],
isValid: MigrationDeps['isValidProfileName']
): string | null | undefined {
let raw: string
try {
raw = readFile(legacyActivePath, 'utf8')
} catch {
return null
}
const name = raw.trim()
if (!name) {
return null
}
if (name === 'default') {
return undefined
}
return isValid(name) ? name : undefined
}
/**
* Return the profile names whose gateway.pid file points to a live hermes process.
* Tolerates missing/malformed pid files and stale-but-recycled PIDs (the latter is
* the whole reason we check both liveness AND cmdline identity).
*
* `hermesHome` is optional so existing call sites that only pass `profilesRoot`
* still work: it is derived as the parent of `…/profiles`.
*/
export function findRunningGatewayProfiles(
profilesRoot: string,
allProfiles: string[],
deps: Pick<MigrationDeps, 'existsSync' | 'readFileSync' | 'isHermesProcess'> & { hermesHome?: string }
): string[] {
const hermesHome = resolveHermesHome(profilesRoot, deps.hermesHome)
const running: string[] = []
for (const name of allProfiles) {
const pidFile = profileGatewayPidPath(name, hermesHome, profilesRoot)
if (!deps.existsSync(pidFile)) {
continue
}
let parsed: { pid?: unknown } | null = null
try {
parsed = JSON.parse(deps.readFileSync(pidFile, 'utf8'))
} catch {
continue
}
const pid = Number(parsed?.pid)
if (!Number.isInteger(pid) || pid < 1) {
continue
}
if (deps.isHermesProcess(pid)) {
running.push(name)
}
}
return running
}
/**
* Hybrid recency × size score for a state.db file. Returns null when the file is
* missing. The formula picks the primary workspace across profiles whose databases
* have been touched at similar times — a 409 MB DB beats a 28 MB one even with a
* slightly newer mtime. Floors the recency weight at 0.1 so a profile touched
* years ago but never deleted still scores > 0 if its DB is large.
*/
export function scoreStateDb(dbPath: string, now: number, stat: MigrationDeps['statSync']): number | null {
let s: { size: number; mtimeMs: number }
try {
s = stat(dbPath)
} catch {
return null
}
const daysSinceModified = Math.max(0, (now - s.mtimeMs) / (1000 * 60 * 60 * 24))
const recencyWeight = Math.max(0.1, 30 - daysSinceModified)
const sizeWeight = Math.log10(Math.max(PROFILE_SCORE_MIN_SIZE_BYTES, s.size))
return recencyWeight * sizeWeight
}
/**
* Pure decision logic. Returns null when nothing migratable was found.
* - legacy non-null → always prefer (no _migrated flag, user-chosen)
* - running.length === 1 → prefer that profile (no flag, gateway-owned)
* - state.db heuristic → set _migrated=true (auto-detected)
* - best === 'default' → suppress write (single-profile fallback)
*/
export function decideMigration(
legacyActive: string | null | undefined,
running: string[],
candidates: string[],
deps: MigrationDeps,
score: (dbPath: string) => number | null
): MigrationDecision | null {
if (legacyActive) {
return { profile: legacyActive }
}
if (running.length === 1) {
return { profile: running[0] }
}
let best: string | null = null
let maxScore = -Infinity
for (const name of candidates) {
const s = score(profileStateDbPath(name, deps.hermesHome, deps.profilesRoot))
if (s == null) {
continue
}
if (s > maxScore) {
maxScore = s
best = name
}
}
if (!best || best === 'default') {
return null
}
return { profile: best, _migrated: true }
}
/**
* List named profile directory names under `profilesRoot`. A directory named
* `default` is accepted if present (unusual) but production default is not a
* child of this folder — see `withDefaultCandidate`.
*/
export function listProfileDirs(deps: MigrationDeps): string[] {
let entries: Dirent[]
try {
entries = deps.readdirSync(deps.profilesRoot, { withFileTypes: true })
} catch {
return []
}
return entries
.filter(e => e.isDirectory() && (e.name === 'default' || deps.isValidProfileName(e.name)))
.map(e => e.name)
}
/** Default is always a candidate; it is `$HERMES_HOME`, not `$HERMES_HOME/profiles/default`. */
export function withDefaultCandidate(named: string[]): string[] {
return ['default', ...named.filter(name => name !== 'default')]
}
/**
* Read an existing active-profile.json. Returns null when missing/malformed.
* `_migrated: true` means the first-boot heuristic wrote it (safe to re-score).
* Absence of that flag is a user/CLI choice and must not be overwritten.
*/
export function readExistingPreference(
desktopProfileConfigPath: string,
readFile: MigrationDeps['readFileSync']
): { profile: string | null; migrated: boolean } | null {
let parsed: unknown
try {
parsed = JSON.parse(readFile(desktopProfileConfigPath, 'utf8'))
} catch {
return null
}
if (!parsed || typeof parsed !== 'object') {
return null
}
const rec = parsed as { profile?: unknown; _migrated?: unknown }
const raw = typeof rec.profile === 'string' ? rec.profile.trim() : ''
return {
profile: raw || null,
migrated: rec._migrated === true
}
}
/**
* First-boot seed, plus repair of heuristic-owned files (`_migrated: true`).
* User-selected files (no `_migrated`) are never overwritten. When a repaired
* heuristic would now pick default, write `{ profile: null }` so Desktop drops
* `--profile` instead of pinning `default`.
*/
export function migrateActiveProfileIfMissing(desktopProfileConfigPath: string, deps: MigrationDeps): boolean {
const existing = deps.existsSync(desktopProfileConfigPath)
? readExistingPreference(desktopProfileConfigPath, deps.readFileSync)
: null
if (existing && !existing.migrated) {
return false
}
const legacyActive = readLegacyActiveProfile(deps.legacyActivePath, deps.readFileSync, deps.isValidProfileName)
const allProfiles = withDefaultCandidate(listProfileDirs(deps))
const running = findRunningGatewayProfiles(deps.profilesRoot, allProfiles, deps)
const candidates = running.length > 1 ? running : allProfiles
const decision = decideMigration(legacyActive, running, candidates, deps, dbPath =>
scoreStateDb(dbPath, deps.now(), deps.statSync)
)
// Same as the heuristic rung: pinning `default` into active-profile.json
// launches `hermes --profile default` and is worse than writing nothing
// (legacy sticky / implicit default). Covers a lone default gateway.pid.
if (!decision || decision.profile === 'default') {
if (existing?.migrated) {
deps.writeJson(desktopProfileConfigPath, { profile: null })
return true
}
return false
}
if (existing?.migrated && existing.profile === decision.profile) {
return false
}
deps.writeJson(desktopProfileConfigPath, decision)
return true
}
@@ -0,0 +1,37 @@
import fs from 'node:fs'
import path from 'node:path'
import { fileURLToPath } from 'node:url'
import { describe, expect, it } from 'vitest'
import { pathForRegistryBackendRequest } from './connection-config'
const here = path.dirname(fileURLToPath(import.meta.url))
const mainSource = fs.readFileSync(path.join(here, 'main.ts'), 'utf8').replace(/\r\n/g, '\n')
describe('primary-remote descriptor reuse keeps profile scope', () => {
it('scopes a shared-remote request with ?profile=<profile>', () => {
expect(pathForRegistryBackendRequest('/api/skills', 'acme', { sharedRemote: true })).toBe(
'/api/skills?profile=acme'
)
})
it('does not add a profile query when the backend is not shared-remote', () => {
// An isolated backend owns one profile; the router must not invent a scope.
expect(pathForRegistryBackendRequest('/api/skills', 'acme', { sharedRemote: false, remoteProfile: null })).toBe(
'/api/skills'
)
})
it('marks the reused primary-remote descriptor sharedRemote so the router scopes it', () => {
const branchStart = mainSource.indexOf("if (id === registry.primary && source.kind !== 'local'")
expect(branchStart).toBeGreaterThan(-1)
const branch = mainSource.slice(branchStart, branchStart + 800)
// The reuse branch must decorate the ambient primary descriptor with
// sharedRemote: true, matching the explicit shared-remote connection path.
expect(branch).toContain('const primaryDescriptor = await ensureBackend(profile)')
expect(branch).toContain('registrySourceOwnsPrimaryBackend(registry, id, primaryDescriptor)')
expect(branch).toContain('sharedRemote: true')
})
})
+130 -7
View File
@@ -619,12 +619,12 @@ test('no-mux: an unrelated listener cannot mask a delayed bind failure', async (
srv.close()
})
test('no-mux: tunnel death after readiness makes the connection unhealthy', async () => {
test('no-mux: tunnel death after readiness triggers a bounded restart, then unhealthy', async () => {
const net = await import('node:net')
const srv = net.createServer()
await new Promise<void>(resolve => srv.listen(0, '127.0.0.1', resolve))
const localPort = (srv.address() as any).port
let tunnel
const tunnels: any[] = []
const spawnFn: any = (_cmd, args) => {
const child: any = new EventEmitter()
@@ -632,10 +632,15 @@ test('no-mux: tunnel death after readiness makes the connection unhealthy', asyn
child.stderr = new EventEmitter()
child.exitCode = null
child.kill = () => {}
child.kill = () => {
child.exitCode = 0
process.nextTick(() => child.emit('exit', 0))
return true
}
if (args.includes('-N')) {
tunnel = child
tunnels.push(child)
process.nextTick(() =>
child.stderr.emit('data', Buffer.from(`Local forwarding listening on 127.0.0.1 port ${localPort}.`))
)
@@ -646,11 +651,129 @@ test('no-mux: tunnel death after readiness makes the connection unhealthy', asyn
return child
}
const conn = new SshConnection({ host: 'box' }, { spawnFn, mux: false })
const conn = new SshConnection(
{ host: 'box' },
{ spawnFn, mux: false, tunnelRestartLimit: 1, tunnelRestartDelayMs: 5 }
)
await conn.open()
await conn.forward(localPort, 9119)
tunnel.emit('exit', 255)
assert.equal(await conn.isAlive(), false)
assert.equal(tunnels.length, 1)
// First death after readiness: a restart is pending, so the connection is
// NOT reported dead — the exact flap that used to cascade into a SIGTERM of
// a healthy backend (#96266).
tunnels[0].exitCode = 255
tunnels[0].emit('exit', 255)
assert.equal(await conn.isAlive(), true, 'flap within the restart budget must not poison isAlive')
// The restart spawns a replacement -N child.
await new Promise(resolve => setTimeout(resolve, 30))
assert.equal(tunnels.length, 2, 'a replacement tunnel child is spawned')
assert.equal(await conn.isAlive(), true)
// Second death exhausts the budget (limit 1): now the connection is dead.
tunnels[1].exitCode = 255
tunnels[1].emit('exit', 255)
await new Promise(resolve => setTimeout(resolve, 30))
assert.equal(tunnels.length, 2, 'no restart past the budget')
assert.equal(await conn.isAlive(), false, 'exhausted budget reports the connection dead')
srv.close()
})
test('no-mux: cancelForward during a pending tunnel restart cancels the restart', async () => {
const net = await import('node:net')
const srv = net.createServer()
await new Promise<void>(resolve => srv.listen(0, '127.0.0.1', resolve))
const localPort = (srv.address() as any).port
const tunnels: any[] = []
const spawnFn: any = (_cmd, args) => {
const child: any = new EventEmitter()
child.stdout = new EventEmitter()
child.stderr = new EventEmitter()
child.exitCode = null
child.kill = () => {
child.exitCode = 0
process.nextTick(() => child.emit('exit', 0))
return true
}
if (args.includes('-N')) {
tunnels.push(child)
process.nextTick(() =>
child.stderr.emit('data', Buffer.from(`Local forwarding listening on 127.0.0.1 port ${localPort}.`))
)
} else {
process.nextTick(() => child.emit('close', 0))
}
return child
}
const conn = new SshConnection(
{ host: 'box' },
{ spawnFn, mux: false, tunnelRestartLimit: 5, tunnelRestartDelayMs: 20 }
)
await conn.forward(localPort, 9119)
tunnels[0].exitCode = 255
tunnels[0].emit('exit', 255)
// Deliberate teardown while the restart timer is pending.
await conn.cancelForward(localPort, 9119)
await new Promise(resolve => setTimeout(resolve, 60))
assert.equal(tunnels.length, 1, 'cancelled forward must not restart its tunnel')
srv.close()
})
test('no-mux: close() during a pending tunnel restart cancels the restart', async () => {
const net = await import('node:net')
const srv = net.createServer()
await new Promise<void>(resolve => srv.listen(0, '127.0.0.1', resolve))
const localPort = (srv.address() as any).port
const tunnels: any[] = []
const spawnFn: any = (_cmd, args) => {
const child: any = new EventEmitter()
child.stdout = new EventEmitter()
child.stderr = new EventEmitter()
child.exitCode = null
child.kill = () => {
child.exitCode = 0
process.nextTick(() => child.emit('exit', 0))
return true
}
if (args.includes('-N')) {
tunnels.push(child)
process.nextTick(() =>
child.stderr.emit('data', Buffer.from(`Local forwarding listening on 127.0.0.1 port ${localPort}.`))
)
} else {
process.nextTick(() => child.emit('close', 0))
}
return child
}
const conn = new SshConnection(
{ host: 'box' },
{ spawnFn, mux: false, tunnelRestartLimit: 5, tunnelRestartDelayMs: 20 }
)
await conn.open()
await conn.forward(localPort, 9119)
tunnels[0].exitCode = 255
tunnels[0].emit('exit', 255)
await conn.close()
await new Promise(resolve => setTimeout(resolve, 60))
assert.equal(tunnels.length, 1, 'closed connection must not restart its tunnels')
srv.close()
})
+174 -54
View File
@@ -39,6 +39,14 @@ import path from 'node:path'
const DEFAULT_CONNECT_TIMEOUT_MS = 15_000
const DEFAULT_EXEC_TIMEOUT_MS = 20_000
const DEFAULT_FORWARD_TIMEOUT_MS = 15_000
// No-mux tunnels are one `ssh -N -L` child each; a transient child death
// (network blip, sshd restart, laptop resume) used to instantly poison
// isAlive() and cascade upstream into a full teardown that SIGTERM'd a
// healthy backend (#96266). Instead, restart the child a bounded number of
// times; consecutive pre-readiness failures exhaust the budget and only then
// is the connection reported dead.
const DEFAULT_TUNNEL_RESTART_LIMIT = 5
const DEFAULT_TUNNEL_RESTART_DELAY_MS = 1_000
const CONTROL_PERSIST_SECONDS = 300
// eslint-disable-next-line no-control-regex -- deliberately reject control chars in ssh targets
@@ -425,7 +433,7 @@ function runSsh(args, { timeoutMs, spawnFn = spawn, stdin = 'ignore', stdinData,
let stderr = ''
let settled = false
const timer = setTimeout(() => {
const timer: any = setTimeout(() => {
if (settled) {
return
}
@@ -547,6 +555,8 @@ class SshConnection {
_connectTimeoutMs: number
_execTimeoutMs: number
_forwardTimeoutMs: number
_tunnelRestartLimit: number
_tunnelRestartDelayMs: number
_opened: boolean
_mux: boolean
_tunnels: Map<string, any>
@@ -588,6 +598,8 @@ class SshConnection {
this._connectTimeoutMs = opts.connectTimeoutMs ?? DEFAULT_CONNECT_TIMEOUT_MS
this._execTimeoutMs = opts.execTimeoutMs ?? DEFAULT_EXEC_TIMEOUT_MS
this._forwardTimeoutMs = opts.forwardTimeoutMs ?? DEFAULT_FORWARD_TIMEOUT_MS
this._tunnelRestartLimit = opts.tunnelRestartLimit ?? DEFAULT_TUNNEL_RESTART_LIMIT
this._tunnelRestartDelayMs = opts.tunnelRestartDelayMs ?? DEFAULT_TUNNEL_RESTART_DELAY_MS
this._opened = false
}
@@ -795,10 +807,149 @@ class SshConnection {
return result.stdout
}
// Spawn one persistent `ssh -N -L` child for a no-mux tunnel and wait for it
// to confirm local forwarding on stderr. Resolves once ready. Rejects on a
// pre-readiness death or confirmation timeout, with the captured stderr on
// `error.tunnelStderr` so the caller can classify auth/bind failures. After
// readiness, a child death is a tunnel FLAP: it is routed into
// _handleNoMuxTunnelFlap (bounded restart) instead of poisoning isAlive()
// outright — the old instant-poison path is how a ~10s local tunnel blip
// cascaded into SIGTERM of a healthy backend (#96266).
_startNoMuxTunnelChild(tunnel: any, spec: string, args: string[], localPort: number | string) {
return new Promise<void>((resolve, reject) => {
const child = this._spawnFn('ssh', args, { stdio: ['ignore', 'ignore', 'pipe'] })
tunnel.child = child
let stderr = ''
let readyConfirmed = false
let settled = false
let downHandled = false
const readyTimeout: any = setTimeout(() => {
finishFail(new Error('tunnel did not confirm local forwarding'))
}, this._forwardTimeoutMs)
readyTimeout.unref?.()
const finishOk = () => {
if (settled) {
return
}
settled = true
clearTimeout(readyTimeout)
resolve()
}
const finishFail = (error: any) => {
if (settled) {
return
}
settled = true
clearTimeout(readyTimeout)
error.tunnelStderr = stderr
reject(error)
}
const onDown = (cause: string, error: any) => {
if (!readyConfirmed) {
tunnel.alive = tunnel.child === child ? false : tunnel.alive
finishFail(error)
return
}
if (downHandled || tunnel.child !== child) {
return
}
downHandled = true
this._handleNoMuxTunnelFlap(tunnel, spec, args, localPort, cause)
}
const readyPattern = new RegExp(`Local forwarding listening on .* port ${localPort}\\b`)
child.stderr?.on('data', (d: any) => {
if (readyConfirmed) {
return
}
stderr = `${stderr}${String(d)}`.slice(-16_384)
if (readyPattern.test(stderr)) {
readyConfirmed = true
finishOk()
}
})
child.on('error', (error: any) => onDown(`tunnel process failed (${error?.message || error})`, error))
child.on('exit', (code: any) =>
onDown(`tunnel process exited with code ${code}`, new Error(`tunnel process exited with code ${code}`))
)
child.on('close', (code: any) =>
onDown(`tunnel process closed with code ${code}`, new Error(`tunnel process closed with code ${code}`))
)
})
}
// A ready no-mux tunnel child died. Deliberate teardown (cancelForward /
// close) and superseded tunnels stay dead; otherwise restart the child up to
// the bounded budget, and only mark the tunnel (and thus the connection)
// unhealthy once the budget is exhausted. The budget is cumulative per
// forward — a tunnel that keeps dying immediately after confirming readiness
// must not restart forever.
_handleNoMuxTunnelFlap(tunnel: any, spec: string, args: string[], localPort: number | string, cause: string) {
if (tunnel.stopping || this._tunnels.get(spec) !== tunnel) {
tunnel.alive = false
return
}
if (tunnel.restarts >= this._tunnelRestartLimit) {
tunnel.alive = false
this._logLine(
`tunnel 127.0.0.1:${localPort} down (${cause}); restart budget exhausted (${this._tunnelRestartLimit})`
)
return
}
tunnel.restarts += 1
this._logLine(
`tunnel 127.0.0.1:${localPort} flapped (${cause}); restarting ` +
`(${tunnel.restarts}/${this._tunnelRestartLimit}) in ${this._tunnelRestartDelayMs}ms`
)
const timer: any = setTimeout(() => {
tunnel.restartTimer = null
if (tunnel.stopping || this._tunnels.get(spec) !== tunnel) {
tunnel.alive = false
return
}
this._startNoMuxTunnelChild(tunnel, spec, args, localPort).then(
() => {
tunnel.alive = true
this._logLine(`tunnel 127.0.0.1:${localPort} restarted`)
},
(error: any) => {
// A restart that never confirmed readiness may leave its child
// running; stop it before deciding whether to retry.
void Promise.resolve(stopTunnelChild(tunnel.child)).catch(() => undefined)
this._handleNoMuxTunnelFlap(tunnel, spec, args, localPort, `restart failed: ${error?.message || error}`)
}
)
}, this._tunnelRestartDelayMs)
timer.unref?.()
tunnel.restartTimer = timer
}
// Establish a local→remote forward. Mux: `-O forward` against the master.
// No-mux: spawn a persistent `ssh -N -L` child that IS the tunnel; ready when
// the local port accepts. The child dying = tunnel down (isAlive of the
// backend catches it upstream).
// the local port accepts. A child dying AFTER readiness is a tunnel flap and
// is restarted with a bounded budget (#96266); only an exhausted budget (or
// a deliberate cancel/close) marks the connection unhealthy for isAlive().
async forward(localPort, remotePort, remoteHost = '127.0.0.1') {
const spec = forwardSpec(localPort, remotePort, remoteHost)
this._logLine(`forwarding 127.0.0.1:${localPort} -> ${remoteHost}:${remotePort}`)
@@ -815,67 +966,20 @@ class SshConnection {
target(this.user, this.host)
]
const child = this._spawnFn('ssh', args, { stdio: ['ignore', 'ignore', 'pipe'] })
const tunnel = { child, alive: true }
const tunnel: any = { alive: true, child: null, restarts: 0, restartTimer: null, stopping: false }
this._tunnels.set(spec, tunnel)
let stderr = ''
let readyConfirmed = false
let readyResolve
let readyReject
const ready = new Promise<void>((resolve, reject) => {
readyResolve = resolve
readyReject = reject
})
const readyPattern = new RegExp(`Local forwarding listening on .* port ${localPort}\\b`)
child.stderr?.on('data', d => {
if (readyConfirmed) {
return
}
stderr = `${stderr}${String(d)}`.slice(-16_384)
if (readyPattern.test(stderr)) {
readyConfirmed = true
readyResolve()
}
})
child.on('error', error => {
tunnel.alive = false
readyReject(error)
})
child.on('exit', code => {
tunnel.alive = false
readyReject(new Error(`tunnel process exited with code ${code}`))
})
child.on('close', code => {
tunnel.alive = false
readyReject(new Error(`tunnel process closed with code ${code}`))
})
let readyTimeout
try {
await Promise.race([
ready,
new Promise((_, reject) => {
readyTimeout = setTimeout(
() => reject(new Error('tunnel did not confirm local forwarding')),
this._forwardTimeoutMs
)
})
])
await this._startNoMuxTunnelChild(tunnel, spec, args, localPort)
} catch (error: any) {
try {
await stopTunnelChild(child)
await stopTunnelChild(tunnel.child)
this._tunnels.delete(spec)
} catch (stopError) {
throw this._fail(stopError, SSH_ERROR.UNKNOWN)
}
throw this._fail(stderr || error, SSH_ERROR.UNKNOWN)
} finally {
clearTimeout(readyTimeout)
throw this._fail(error?.tunnelStderr || error, SSH_ERROR.UNKNOWN)
}
return
@@ -904,6 +1008,14 @@ class SshConnection {
const tunnel = this._tunnels.get(spec)
if (tunnel) {
tunnel.stopping = true
tunnel.alive = false
if (tunnel.restartTimer) {
clearTimeout(tunnel.restartTimer)
tunnel.restartTimer = null
}
await stopTunnelChild(tunnel.child)
this._tunnels.delete(spec)
this._logLine(`cancelled forward 127.0.0.1:${localPort}`)
@@ -931,6 +1043,14 @@ class SshConnection {
if (!this._mux) {
for (const [spec, tunnel] of this._tunnels) {
tunnel.stopping = true
tunnel.alive = false
if (tunnel.restartTimer) {
clearTimeout(tunnel.restartTimer)
tunnel.restartTimer = null
}
await stopTunnelChild(tunnel.child)
this._tunnels.delete(spec)
}
+28 -9
View File
@@ -13,6 +13,7 @@ import nodePty from 'node-pty'
import { resolveTerminalConnectionForSender } from './connection-apply'
import { ensureSpawnHelperExecutable } from './spawn-helper-perms'
import { buildInteractiveSshArgs } from './ssh-connection'
import { createTerminalOutputGate } from './terminal-output-gate'
import { buildWindowsInteractiveCommand } from './windows-remote-lifecycle'
export interface TerminalIpcDeps {
@@ -312,12 +313,6 @@ export function registerTerminalIpc({
)
: nodePty.spawn(command, args, { cols, cwd, env: terminalShellEnv(), name: 'xterm-256color', rows })
terminalSessions.set(id, {
pty: ptyProcess,
webContentsId: event.sender.id,
...(remote ? { sshScope: sshTarget.scope, remoteCwd: String(payload?.cwd || '') } : {})
})
const send = (suffix, payload) => {
if (event.sender.isDestroyed()) {
return
@@ -326,16 +321,40 @@ export function registerTerminalIpc({
event.sender.send(terminalChannel(id, suffix), payload)
}
ptyProcess.onData(data => send('data', data))
const outputGate = createTerminalOutputGate({
onExitFlushed: () => terminalSessions.delete(id),
sendData: data => send('data', data),
sendExit: payload => send('exit', payload)
})
terminalSessions.set(id, {
outputGate,
pty: ptyProcess,
webContentsId: event.sender.id,
...(remote ? { sshScope: sshTarget.scope, remoteCwd: String(payload?.cwd || '') } : {})
})
ptyProcess.onData(data => outputGate.data(data))
ptyProcess.onExit(({ exitCode, signal }) => {
terminalSessions.delete(id)
send('exit', { code: exitCode, signal: signal || null })
outputGate.exit({ code: exitCode, signal: signal == null ? null : String(signal) })
})
event.sender.once('destroyed', () => disposeTerminalSession(id))
return { cwd: remote ? null : cwd, id, shell: remote ? 'ssh' : name }
})
ipcMain.handle('hermes:terminal:attach', (event, id) => {
const sessionInfo = terminalSessions.get(String(id || ''))
if (!sessionInfo || sessionInfo.webContentsId !== event.sender.id) {
return false
}
sessionInfo.outputGate.attach()
return true
})
ipcMain.handle('hermes:terminal:write', (_event, id, data) => {
const sessionInfo = terminalSessions.get(String(id || ''))
@@ -0,0 +1,57 @@
import assert from 'node:assert/strict'
import { test } from 'vitest'
import { createTerminalOutputGate } from './terminal-output-gate'
test('holds the initial shell prompt until the renderer attaches', () => {
const events: string[] = []
const gate = createTerminalOutputGate({
onExitFlushed: () => events.push('cleanup'),
sendData: data => events.push(`data:${data}`),
sendExit: payload => events.push(`exit:${payload.code}`)
})
gate.data('$ ')
assert.deepEqual(events, [])
gate.attach()
assert.deepEqual(events, ['data:$ '])
gate.data('pwd\r\n')
assert.deepEqual(events, ['data:$ ', 'data:pwd\r\n'])
})
test('flushes buffered output before an early shell exit', () => {
const events: string[] = []
const gate = createTerminalOutputGate({
onExitFlushed: () => events.push('cleanup'),
sendData: data => events.push(`data:${data}`),
sendExit: payload => events.push(`exit:${payload.code}:${payload.signal}`)
})
gate.data('startup failed\r\n')
gate.exit({ code: 1, signal: null })
assert.deepEqual(events, [])
gate.attach()
assert.deepEqual(events, ['data:startup failed\r\n', 'exit:1:null', 'cleanup'])
})
test('attach is idempotent and never replays startup output twice', () => {
const events: string[] = []
const gate = createTerminalOutputGate({
onExitFlushed: () => events.push('cleanup'),
sendData: data => events.push(`data:${data}`),
sendExit: payload => events.push(`exit:${payload.code}`)
})
gate.data('$ ')
gate.attach()
gate.attach()
assert.deepEqual(events, ['data:$ '])
})
@@ -0,0 +1,62 @@
export interface TerminalExitPayload {
code: number | null
signal: string | null
}
interface TerminalOutputGateOptions {
onExitFlushed: () => void
sendData: (data: string) => void
sendExit: (payload: TerminalExitPayload) => void
}
const MAX_PENDING_OUTPUT = 1024 * 1024
export function createTerminalOutputGate({ onExitFlushed, sendData, sendExit }: TerminalOutputGateOptions) {
let attached = false
let pendingData = ''
let pendingExit: TerminalExitPayload | null = null
const flushExit = (payload: TerminalExitPayload) => {
sendExit(payload)
onExitFlushed()
}
return {
attach(): void {
if (attached) {
return
}
attached = true
if (pendingData) {
sendData(pendingData)
pendingData = ''
}
if (pendingExit) {
const payload = pendingExit
pendingExit = null
flushExit(payload)
}
},
data(chunk: string): void {
if (attached) {
sendData(chunk)
return
}
pendingData = `${pendingData}${chunk}`.slice(-MAX_PENDING_OUTPUT)
},
exit(payload: TerminalExitPayload): void {
if (attached) {
flushExit(payload)
return
}
pendingExit = payload
}
}
}
@@ -75,6 +75,12 @@ describe('formatProbeFailedMessage', () => {
assert.ok(msg.includes('hermes update'))
assert.ok(msg.includes('retry'))
})
it('distinguishes a timeout from a confirmed blocker', () => {
const msg = formatProbeFailedMessage('timed out after 60 seconds')
assert.ok(msg.includes('timed out after 60 seconds'))
assert.ok(msg.includes('no blocking process was confirmed'))
})
})
// ---------------------------------------------------------------------------
@@ -100,6 +106,45 @@ describe('parseVenvBlockerScanOutput', () => {
assert.equal(o.kind, 'blocked')
})
// Contract fixture (#98336/#98350): the scanner reports exemption
// diagnostics (counts + sanitized evidence) alongside the authoritative
// blocked/processes fields. The consumer must tolerate those fields today
// and must keep enforcing blocked/processes consistency — a future parser
// change that either chokes on the diagnostics or silently reinterprets
// an exemption as a blocker breaks this fixture.
it('tolerates exemption diagnostics while enforcing blocked/processes consistency', () => {
const clear = parseVenvBlockerScanOutput(
ok({
pausable_gateways: 2,
deferred_backends: 1,
deferred_backend_evidence: [{ pid: 78, purpose: 'serve', port: 9119 }]
})
)
assert.equal(clear.kind, 'clear')
const blocked = parseVenvBlockerScanOutput(
ok({
blocked: true,
processes: [{ pid: 79, name: 'python.exe', cmdline: 'c' }],
pausable_gateways: 1,
deferred_backends: 1,
deferred_backend_evidence: [{ pid: 78, purpose: 'serve', port: 9119 }]
})
)
assert.equal(blocked.kind, 'blocked')
if (blocked.kind !== 'blocked') {
return
}
assert.deepEqual(
blocked.result.processes.map(p => p.pid),
[79]
)
})
it('classifies Python http.server blockers as safe local previews with a human label', () => {
const o = parseVenvBlockerScanOutput(
ok({
@@ -254,6 +299,15 @@ describe('scanVenvBlockers', () => {
}) as any
}
function execTimeout(): any {
return (async (...args: any[]) => {
const e: any = new Error()
e.killed = true
e.signal = 'SIGTERM'
throw e
}) as any
}
it('clear scan returns clear', async () => {
assert.equal((await scanVenvBlockers('/r', execReturn(okJson), stubVenv)).kind, 'clear')
})
@@ -267,6 +321,14 @@ describe('scanVenvBlockers', () => {
assert.equal(o.kind, 'probe-failure')
})
it('reports a timed-out subprocess explicitly', async () => {
const o = await scanVenvBlockers('/r', execTimeout(), stubVenv)
assert.deepEqual(o, {
kind: 'probe-failure',
error: 'timed out after 60 seconds'
})
})
it('missing venv python is probe-failure', async () => {
const o = await scanVenvBlockers('/r', execReturn(okJson), () => null)
assert.equal(o.kind, 'probe-failure')
@@ -292,8 +354,7 @@ describe('scanVenvBlockers', () => {
assert.ok(c.cmd.endsWith('python.exe'))
assert.deepEqual(c.args, ['-m', 'hermes_cli._scan_venv_blockers'])
assert.equal(c.cwd, '/update/root')
assert.equal(typeof c.timeout, 'number')
assert.ok(c.timeout > 0)
assert.equal(c.timeout, 60_000)
})
})
+20 -5
View File
@@ -45,7 +45,10 @@ export type ScanOutcome =
// Constants
// ---------------------------------------------------------------------------
const SCAN_TIMEOUT_MS = 15000
// A Windows process table can take longer than 15 seconds to inspect on busy
// hosts. Keep a watchdog for genuinely wedged probes, but leave enough headroom
// for the scanner's conservative fallback checks.
const SCAN_TIMEOUT_MS = 60000
const SCAN_MODULE = 'hermes_cli._scan_venv_blockers'
// ---------------------------------------------------------------------------
@@ -204,7 +207,7 @@ export function parseVenvBlockerScanOutput(raw: string): ScanOutcome {
/**
* Run the venv-blocker scan subprocess. Async so the Electron main-process
* event loop is never blocked by the psutil process scan (up to 15s on a
* event loop is never blocked by the psutil process scan (tens of seconds on a
* loaded Windows box). Accepts optional overrides for testing (dependency
* injection).
*/
@@ -233,6 +236,13 @@ export async function scanVenvBlockers(
stdout = String((proc as any).stdout ?? '')
} catch (err: any) {
if (err?.killed === true) {
return {
kind: 'probe-failure',
error: `timed out after ${SCAN_TIMEOUT_MS / 1000} seconds`
}
}
const diag = [`exit code ${err.status ?? err.code ?? -1}`]
if (err.stderr) {
@@ -298,10 +308,15 @@ export function formatBlockerMessage(result: VenvBlockerScanResult): string {
/**
* Build a probe-failure error message.
*/
export function formatProbeFailedMessage(): string {
export function formatProbeFailedMessage(error?: string): string {
const timeoutDetail = error?.startsWith('timed out after')
? `\n\nThe verification scan ${error}; no blocking process was confirmed.`
: ''
return (
'Update aborted: Desktop could not verify the Hermes installation is free.\n' +
'\n' +
'Update aborted: Desktop could not verify the Hermes installation is free.' +
timeoutDetail +
'\n\n' +
'Close other Hermes windows and terminals, then retry. If the problem\n' +
'persists, run `hermes update` in a terminal for detailed diagnostics.'
)
@@ -0,0 +1,57 @@
import assert from 'node:assert/strict'
import { test } from 'vitest'
import { hasWindowsPathPrefix, isHermesOwnedVenvDaemon } from './venv-holder-select'
const SCRIPTS = 'C:\\Hermes\\venv\\Scripts'
test('matches the hindsight daemon shim (exe under venv Scripts + hindsight cmdline)', () => {
assert.equal(
isHermesOwnedVenvDaemon(
'C:\\Hermes\\venv\\Scripts\\pythonw.exe',
'C:\\Hermes\\venv\\Scripts\\pythonw.exe -m hindsight_api.main --daemon --idle-timeout 300 --port 9177',
SCRIPTS
),
true
)
})
test('Windows path prefix match is ordinal case-insensitive', () => {
assert.equal(
isHermesOwnedVenvDaemon(
'c:\\hermes\\venv\\scripts\\python.exe',
'python.exe -m hindsight_api.main --daemon',
'C:\\Hermes\\venv\\Scripts'
),
true
)
})
test('excludes external venv holders that are not the hindsight daemon', () => {
// a user terminal running the hermes CLI from the venv — must NOT be killed
assert.equal(isHermesOwnedVenvDaemon('C:\\Hermes\\venv\\Scripts\\hermes.exe', 'hermes chat -q "hi"', SCRIPTS), false)
// an unrelated python script using the venv interpreter
assert.equal(
isHermesOwnedVenvDaemon('C:\\Hermes\\venv\\Scripts\\python.exe', 'python C:\\tools\\import.py', SCRIPTS),
false
)
})
test('excludes exes outside the venv even when the cmdline mentions hindsight', () => {
assert.equal(
isHermesOwnedVenvDaemon('C:\\Other\\pythonw.exe', 'pythonw -m hindsight_api.main --daemon', SCRIPTS),
false
)
})
test('prefix boundary: sibling dirs (ScriptsX) do not match', () => {
assert.equal(hasWindowsPathPrefix('C:\\Hermes\\venv\\ScriptsX\\python.exe', SCRIPTS), false)
assert.equal(hasWindowsPathPrefix('C:\\Hermes\\venv\\Scripts\\python.exe', SCRIPTS), true)
})
test('null/undefined fields never match', () => {
assert.equal(isHermesOwnedVenvDaemon(null, 'x', SCRIPTS), false)
assert.equal(isHermesOwnedVenvDaemon('C:\\Hermes\\venv\\Scripts\\pythonw.exe', null, SCRIPTS), false)
assert.equal(isHermesOwnedVenvDaemon(undefined, undefined, SCRIPTS), false)
})
@@ -0,0 +1,36 @@
/**
* venv-holder-select.ts
*
* Pure Windows venv-holder selection logic (testable without Electron).
*
* The pre-update handoff kills Hermes-OWNED venv daemons (the memory plugin's
* hindsight daemon) so the updater never races a mapped shim. External
* holders (a user terminal running `hermes`, unrelated scripts) must NOT be
* killed — current design reports them via scanVenvBlockers and ABORTS the
* handoff instead (main.ts releaseBackendLock / applyUpdates).
*/
/** Ordinal case-insensitive prefix check for Windows paths. */
export function hasWindowsPathPrefix(exePath: string, venvScriptsDir: string): boolean {
const prefix = `${venvScriptsDir}\\`
return exePath.length >= prefix.length && exePath.slice(0, prefix.length).toLowerCase() === prefix.toLowerCase()
}
/**
* True when a process is a Hermes-owned venv daemon: its exe lives under
* `<venv>\Scripts\` (ordinal case-insensitive prefix) AND its cmdline
* references `hindsight_api.main` (the memory daemon the memory plugin
* spawns DETACHED — it outlives Hermes and holds venv shims mapped).
*/
export function isHermesOwnedVenvDaemon(
exePath: string | null | undefined,
cmdline: string | null | undefined,
venvScriptsDir: string
): boolean {
if (!exePath || !cmdline) {
return false
}
return hasWindowsPathPrefix(exePath, venvScriptsDir) && /hindsight_api\.main/i.test(cmdline)
}
+4
View File
@@ -28,6 +28,10 @@ export default defineConfig({
/* Test files live under e2e/ so they never collide with the vitest suite
* under src/ or the node:test files under electron/. */
testDir: './e2e',
/* ...except `*.unit.test.ts`, which covers the e2e HELPERS (no Electron, no
* app) and is owned by the vitest `electron` project. Without this the
* default testMatch would claim those files too and run them twice. */
testIgnore: '**/*.unit.test.ts',
/* The desktop app can take a while to bootstrap on cold CI runners — 90 s
* per test gives us headroom without masking real hangs. */
timeout: 90_000,
@@ -17,6 +17,14 @@ import { isMain } from "./utils.mjs"
const ROUTER_CONTEXT_ERROR = "may be used only in the context of a"
// @tanstack/react-query carries module-level React context (QueryClientContext).
// The entry's QueryClientProvider and every lazy chunk's useQuery must share ONE
// runtime instance; if a build ever emits a second copy, the provider's context
// is invisible to the other copy and useQuery throws "No QueryClient set" — the
// packaged app error-boundaries on launch (#95560). Same single-instance
// invariant as the react-router check above, same failure class.
const QUERY_CLIENT_CONTEXT_ERROR = "No QueryClient set, use QueryClientProvider to set one"
// Pure check — returns { ok: true } or { ok: false, error: "..." }.
// Kept side-effect-free so it can be unit tested without spawning a process.
export function checkDistBuilt(distDir) {
@@ -54,6 +62,21 @@ export function checkDistBuilt(distDir) {
}
}
const queryClientContextAssets = readdirSync(assetsDir)
.filter(name => name.endsWith(".js"))
.filter(name => readFileSync(join(assetsDir, name), "utf8").includes(QUERY_CLIENT_CONTEXT_ERROR))
if (queryClientContextAssets.length > 1) {
return {
ok: false,
error:
`@tanstack/react-query context invariant found in multiple JS assets: ` +
`${queryClientContextAssets.join(", ")} — duplicate react-query runtimes make the ` +
`QueryClientProvider's context invisible to useQuery in other chunks (` +
`"No QueryClient set" on launch, #95560)`
}
}
return { ok: true }
}

Some files were not shown because too many files have changed in this diff Show More